From d0550ed83ac8d3140b3783348d665fe589c01b13 Mon Sep 17 00:00:00 2001 From: Shalom Date: Tue, 8 Sep 2026 18:39:21 +0300 Subject: [PATCH 01/82] fix(search): one search per leg, not one per leg per collection structuredSearch loops the collection list and pushes each collection's results as its own RRF ranked list, so a scope of N collections builds N lists per search leg. RRF ranks with no relevance floor, so every collection's best-of-a-bad-lot enters fusion as a rank 1 beside the real answer. The 2x weight given to rankedLists[0] compounds it: it is meant for a caller who ordered `searches` by importance, but the fan-out makes list 0 whichever collection happened to sort first. The visible symptom is that ranking depends on argument order. Over a 21-collection scope and a 14-question gold set, reversing the collections array moved rank 1 for 13 of 14 queries. searchFTS and searchVec already accept `string | readonly string[]`, so this just passes the scope straight through instead of looping it. On that same set: P@1 0.000 -> 0.500, recall@5 0.643 -> 0.786, MRR 0.279 -> 0.607, and reversing the array no longer changes anything. Median query time 0.499s -> 0.060s, since it is now one FTS and one vector query rather than 21 of each -- searchVec in particular was pulling the same global top-k every time and only then filtering by collection. Single-collection scopes are unchanged, as they must be: the loop ran exactly once there already. (cherry picked from commit aa67e26cef7afdb6bc6020104f53c3a7f1bc399f) --- src/store.ts | 64 +++++++++++++++++++++++++--------------------------- 1 file changed, 31 insertions(+), 33 deletions(-) diff --git a/src/store.ts b/src/store.ts index 43a0cef3b..6ca666011 100644 --- a/src/store.ts +++ b/src/store.ts @@ -6410,26 +6410,26 @@ export async function structuredSearch( `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` ).get(); - // Helper to run search across collections (or all if undefined) - const collectionList = collections ?? [undefined]; // undefined = all collections + // One query for the whole scope. searchFTS/searchVec already take a list, so + // looping the collections here only split one result set into N ranked lists, + // and RRF then treated each collection's best-of-a-bad-lot as a rank 1. + const collScope = collections && collections.length > 0 ? collections : undefined; // Step 1: Run FTS for all lex searches (sync, instant) for (const search of searches) { if (search.type === 'lex') { - for (const coll of collectionList) { - const ftsResults = store.searchFTS(search.query, 20, coll, filter); - if (ftsResults.length > 0) { - for (const r of ftsResults) docidMap.set(r.filepath, r.docid); - rankedLists.push(ftsResults.map(r => ({ - file: r.filepath, displayPath: r.displayPath, - title: r.title, body: r.body || "", score: r.score, - }))); - rankedListMeta.push({ - source: "fts", - queryType: "lex", - query: search.query, - }); - } + const ftsResults = store.searchFTS(search.query, 20, collScope, filter); + if (ftsResults.length > 0) { + for (const r of ftsResults) docidMap.set(r.filepath, r.docid); + rankedLists.push(ftsResults.map(r => ({ + file: r.filepath, displayPath: r.displayPath, + title: r.title, body: r.body || "", score: r.score, + }))); + rankedListMeta.push({ + source: "fts", + queryType: "lex", + query: search.query, + }); } } } @@ -6453,23 +6453,21 @@ export async function structuredSearch( const embedding = embeddings[i]?.embedding; if (!embedding) continue; - for (const coll of collectionList) { - const vecResults = await store.searchVec( - vecSearches[i]!.query, embedModel, 20, coll, - undefined, embedding, filter - ); - if (vecResults.length > 0) { - for (const r of vecResults) docidMap.set(r.filepath, r.docid); - rankedLists.push(vecResults.map(r => ({ - file: r.filepath, displayPath: r.displayPath, - title: r.title, body: r.body || "", score: r.score, - }))); - rankedListMeta.push({ - source: "vec", - queryType: vecSearches[i]!.type, - query: vecSearches[i]!.query, - }); - } + const vecResults = await store.searchVec( + vecSearches[i]!.query, embedModel, 20, collScope, + undefined, embedding, filter + ); + if (vecResults.length > 0) { + for (const r of vecResults) docidMap.set(r.filepath, r.docid); + rankedLists.push(vecResults.map(r => ({ + file: r.filepath, displayPath: r.displayPath, + title: r.title, body: r.body || "", score: r.score, + }))); + rankedListMeta.push({ + source: "vec", + queryType: vecSearches[i]!.type, + query: vecSearches[i]!.query, + }); } } } From c20f733b9c1c8acb8bfc1f33e7221437eb52916b Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 10:53:15 -0500 Subject: [PATCH 02/82] test(search): cover #946's single search per leg over a collection scope #946 carries no tests. These pin its two effects on structuredSearch: - a vec leg over collections [alpha, beta] calls searchVec once with the whole list, where it used to call it once per collection; - collections: [] searches every collection, where the per-collection loop used to run zero times and return nothing. --- test/structured-search.test.ts | 55 +++++++++++++++++++++++++++++++++- 1 file changed, 54 insertions(+), 1 deletion(-) diff --git a/test/structured-search.test.ts b/test/structured-search.test.ts index 70da7fd1e..d216266c1 100644 --- a/test/structured-search.test.ts +++ b/test/structured-search.test.ts @@ -9,7 +9,7 @@ * Run with: bun test structured-search.test.ts */ -import { describe, test, expect, beforeAll, afterAll } from "vitest"; +import { describe, test, expect, vi, beforeAll, afterAll, beforeEach, afterEach } from "vitest"; import { mkdtemp, rm } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; @@ -592,3 +592,56 @@ describe("buildFTS5Query (lex parser)", () => { expect(buildFTS5Query("performance -sports")).toBe('"performance"* NOT "sports"*'); }); }); + +// ============================================================================= +// structuredSearch collection scope +// ============================================================================= + +describe("structuredSearch collection scope", () => { + let testDir: string; + let store: Store; + const origConfigDir = process.env.QMD_CONFIG_DIR; + + beforeEach(async () => { + testDir = await mkdtemp(join(tmpdir(), "qmd-structured-scope-")); + process.env.QMD_CONFIG_DIR = await mkdtemp(join(testDir, "config-")); + store = createStore(join(testDir, "index.sqlite")); + }); + + afterEach(async () => { + store.close(); + if (origConfigDir === undefined) delete process.env.QMD_CONFIG_DIR; + else process.env.QMD_CONFIG_DIR = origConfigDir; + await rm(testDir, { recursive: true, force: true }); + }); + + function addDocument(collection: string, path: string, body: string): void { + const now = new Date().toISOString(); + const hash = `hash-${collection}-${path}`; + store.insertContent(hash, body, now); + store.insertDocument(collection, path, path, hash, now, now); + } + + test("a vec leg over several collections runs one vector search over the whole list", async () => { + store.ensureVecTable(3); + store.llm = { + embedModelName: "scope-embed-model", + embedBatch: async (texts: string[]) => texts.map(() => ({ embedding: [1, 0, 0], model: "scope-embed-model" })), + } as any; + const searchVec = vi.fn(async () => []); + store.searchVec = searchVec as any; + + await structuredSearch(store, [{ type: "vec", query: "x" }], { collections: ["alpha", "beta"], skipRerank: true }); + + expect(searchVec).toHaveBeenCalledTimes(1); + expect(searchVec.mock.calls[0]![3]).toEqual(["alpha", "beta"]); + }); + + test("an empty collection list searches every collection", async () => { + addDocument("alpha", "a.md", "# A\n\nquokkascope in alpha"); + addDocument("beta", "b.md", "# B\n\nquokkascope in beta"); + + const results = await structuredSearch(store, [{ type: "lex", query: "quokkascope" }], { collections: [], skipRerank: true }); + expect(results.map(r => r.displayPath).sort()).toEqual(["alpha/a.md", "beta/b.md"]); + }); +}); From d9832902f07f02b00f0e6735da226cca818ec028 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Christoph=20W=C3=BCbbels?= Date: Mon, 28 Sep 2026 12:33:00 +0200 Subject: [PATCH 03/82] fix(search): fuse one ranked list per search over all named collections structuredSearch ran every lex/vec search once per named collection and fused each per-collection list separately, so every collection's best hit tied for rank 1 and the 2x RRF weight meant for the first search went to the first named collection. Results came back in the order collections were named. searchFTS/searchVec already accept a collection list and merge by score, so pass the list through and build one ranked list per search. (cherry picked from commit 7b203c7f283201e51bb958854dbcb6242e4d5a2a) --- src/store.ts | 11 +++++------ test/structured-search.test.ts | 24 ++++++++++++++++++++++++ 2 files changed, 29 insertions(+), 6 deletions(-) diff --git a/src/store.ts b/src/store.ts index 6ca666011..808cd6b1b 100644 --- a/src/store.ts +++ b/src/store.ts @@ -6410,15 +6410,14 @@ export async function structuredSearch( `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` ).get(); - // One query for the whole scope. searchFTS/searchVec already take a list, so - // looping the collections here only split one result set into N ranked lists, - // and RRF then treated each collection's best-of-a-bad-lot as a rank 1. - const collScope = collections && collections.length > 0 ? collections : undefined; + // Each search yields ONE ranked list over the union of the named collections + // (undefined = all). searchFTS/searchVec merge per-collection results by score, + // so the RRF weight below boosts the first search, not the first collection. // Step 1: Run FTS for all lex searches (sync, instant) for (const search of searches) { if (search.type === 'lex') { - const ftsResults = store.searchFTS(search.query, 20, collScope, filter); + const ftsResults = store.searchFTS(search.query, 20, collections, filter); if (ftsResults.length > 0) { for (const r of ftsResults) docidMap.set(r.filepath, r.docid); rankedLists.push(ftsResults.map(r => ({ @@ -6454,7 +6453,7 @@ export async function structuredSearch( if (!embedding) continue; const vecResults = await store.searchVec( - vecSearches[i]!.query, embedModel, 20, collScope, + vecSearches[i]!.query, embedModel, 20, collections, undefined, embedding, filter ); if (vecResults.length > 0) { diff --git a/test/structured-search.test.ts b/test/structured-search.test.ts index d216266c1..f8c7e9650 100644 --- a/test/structured-search.test.ts +++ b/test/structured-search.test.ts @@ -15,6 +15,9 @@ import { tmpdir } from "node:os"; import { join } from "node:path"; import { createStore, + hashContent, + insertContent, + insertDocument, structuredSearch, validateSemanticQuery, validateLexQuery, @@ -345,6 +348,27 @@ describe("structuredSearch", () => { { type: "lex", query: "\"unfinished phrase", line: 2 } ])).rejects.toThrow(/unmatched double quote/); }); + + test("ranks over the union of named collections, not by the order they are named", async () => { + const now = new Date().toISOString(); + const add = async (collection: string, path: string, body: string) => { + const hash = await hashContent(body); + insertContent(store.db, hash, body, now); + insertDocument(store.db, collection, path, path, hash, now, now); + }; + const filler = "lorem ipsum dolor sit amet consectetur adipiscing elit ".repeat(40); + await add("alpha", "alpha/weak.md", `${filler} zanzibarquartz ${filler}`); + await add("beta", "beta/strong.md", "zanzibarquartz zanzibarquartz zanzibarquartz notes"); + + const searches: ExpandedQuery[] = [{ type: "lex", query: "zanzibarquartz" }]; + const unscoped = await structuredSearch(store, searches, { skipRerank: true }); + expect(unscoped[0]?.displayPath).toContain("strong.md"); + + for (const collections of [["alpha", "beta"], ["beta", "alpha"]]) { + const results = await structuredSearch(store, searches, { collections, skipRerank: true }); + expect(results.map(r => r.displayPath)).toEqual(unscoped.map(r => r.displayPath)); + } + }); }); // ============================================================================= From 176b9b931a09b0f5a6b41fc168dcdde33d5336b4 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 10:54:04 -0500 Subject: [PATCH 04/82] test(search): guard vec-leg ranking against the order collections are named #1009 makes the same change as #946 below it, so no test of it can fail on that layer; this is a guard. #1009's own test covers the keyword leg. This one covers the vector leg: with real vectors in two collections, a vec search scoped to [alpha, beta] or [beta, alpha] returns the unscoped order. --- test/structured-search.test.ts | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/test/structured-search.test.ts b/test/structured-search.test.ts index f8c7e9650..6d116e0f2 100644 --- a/test/structured-search.test.ts +++ b/test/structured-search.test.ts @@ -661,6 +661,33 @@ describe("structuredSearch collection scope", () => { expect(searchVec.mock.calls[0]![3]).toEqual(["alpha", "beta"]); }); + test("a vec leg ranks the same whichever order the collections are named in", async () => { + const embedModel = "scope-embed-model"; + store.ensureVecTable(3); + store.llm = { + embedModelName: embedModel, + embedBatch: async (texts: string[]) => texts.map(() => ({ embedding: [1, 0, 0], model: embedModel })), + } as any; + const now = new Date().toISOString(); + const vectors: [string, string, number[]][] = [ + ["alpha", "far.md", [0, 1, 0]], + ["beta", "near.md", [1, 0, 0]], + ["alpha", "mid.md", [0.8, 0.6, 0]], + ]; + for (const [collection, path, vector] of vectors) { + addDocument(collection, path, `# ${path}\n\nbody`); + store.insertEmbedding(`hash-${collection}-${path}`, 0, 0, new Float32Array(vector), embedModel, now); + } + + const searches: ExpandedQuery[] = [{ type: "vec", query: "x" }]; + const unscoped = await structuredSearch(store, searches, { skipRerank: true }); + expect(unscoped.map(r => r.displayPath)).toEqual(["beta/near.md", "alpha/mid.md", "alpha/far.md"]); + for (const collections of [["alpha", "beta"], ["beta", "alpha"]]) { + const scoped = await structuredSearch(store, searches, { collections, skipRerank: true }); + expect(scoped.map(r => r.displayPath)).toEqual(unscoped.map(r => r.displayPath)); + } + }); + test("an empty collection list searches every collection", async () => { addDocument("alpha", "a.md", "# A\n\nquokkascope in alpha"); addDocument("beta", "b.md", "# B\n\nquokkascope in beta"); From ba2a2db2fb0d0904c51c7af6d9767a7d99d70621 Mon Sep 17 00:00:00 2001 From: Oliver Ratzesberger Date: Sat, 22 Aug 2026 00:30:34 +0200 Subject: [PATCH 05/82] fix(store): scope keyword retrieval at the index, not after it searchFTS() took a global top-(limit * 10) from the FTS5 index and then dropped the rows outside the requested collection. The over-fetch could not make a post-filter correct, it only moved the cutoff: a collection holding a small share of the index loses every row whenever the global window happens to contain none of its documents, and the caller cannot tell that apart from the collection genuinely having nothing on the subject. This is the keyword-side twin of the vector fix in #847, which replaced the same global-then-filter shape in searchVec() with a collection-scoped scan. searchFTS() kept it, softened only by BM25 being far more selective than ANN. The scope now rides in the FTS5 lookup as a rowid prefilter. documents_fts.rowid is documents.id, so the subquery is a covering-index read on idx_documents_collection and the MATCH still runs against the FTS5 index (EXPLAIN QUERY PLAN: SCAN documents_fts VIRTUAL TABLE INDEX 0:=M3), which is what the CTE exists to protect. The limit * 10 over-fetch is gone: the CTE now takes exactly `limit` rows, because every one of them is in scope. Multi-collection scopes are unaffected. They still fan out per collection and merge by score (#871); each leg is now scoped at the index. Test: 40 short noise documents carrying the term in their titles, plus one long target document in a second collection. Scoped to that second collection the target is the only answer. Before this change the same call returned [], because all 20 candidates in the global window belonged to the noise collection. Co-Authored-By: Claude Fable 5 (cherry picked from commit 016e2aa73f4451eb550156dd6f4884a5083178a7) --- src/store.ts | 54 ++++++++++++++++++++++------------------------ test/store.test.ts | 38 ++++++++++++++++++++++++++++++++ 2 files changed, 64 insertions(+), 28 deletions(-) diff --git a/src/store.ts b/src/store.ts index 808cd6b1b..cb37cf977 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4456,27 +4456,37 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle const ftsQuery = buildFTS5Query(query); if (!ftsQuery) return []; - // Use a CTE to force FTS5 to run first, then filter by collection. - // Without the CTE, SQLite's query planner combines FTS5 MATCH with the - // collection filter in a single WHERE clause, which can cause it to - // abandon the FTS5 index and fall back to a full scan — turning an 8ms - // query into a 17-second query on large collections. + // Use a CTE to keep FTS5 driving the query. Without it, SQLite's planner + // combines the FTS5 MATCH and the collection filter into a single WHERE + // clause, which can make it abandon the FTS5 index and fall back to a full + // scan, turning an 8ms query into a 17-second query on large collections. const params: (string | number)[] = [ftsQuery]; - // When filtering by collection or metadata, fetch extra candidates from the - // FTS index since some will be filtered out. Without a filter we can fetch - // exactly the requested limit. Selective filters remain best-effort: an - // eligible document outside this candidate window is missed (same - // completeness contract as collection filtering). - const ftsLimit = (collectionFilter || filter) ? limit * 10 : limit; + // Scope the FTS lookup itself with a rowid prefilter instead of taking a + // global top-N and dropping the out-of-scope rows afterwards. The over-fetch + // this replaces (limit * 10 candidates when scoped) could not make a + // post-filter correct, it only moved the cutoff: a collection holding a + // small share of the index still lost every row whenever the global top-N + // happened to contain none of its documents, and the caller got an empty + // result that reads as "this collection has nothing on the subject". + // + // documents_fts.rowid is documents.id, so the subquery is a covering-index + // read on idx_documents_collection and the MATCH still runs against the FTS5 + // index (EXPLAIN QUERY PLAN: SCAN documents_fts VIRTUAL TABLE INDEX 0:=M3), + // which is what the CTE above exists to protect. + let scopeFilter = ""; + if (collectionFilter) { + scopeFilter = `\n AND rowid IN (SELECT id FROM documents WHERE active = 1 AND collection = ?)`; + params.push(String(collectionFilter)); + } - let sql = ` + const sql = ` WITH fts_matches AS ( SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score FROM documents_fts - WHERE documents_fts MATCH ? + WHERE documents_fts MATCH ?${scopeFilter} ORDER BY bm25_score ASC - LIMIT ${ftsLimit} + LIMIT ${limit} ) SELECT 'qmd://' || d.collection || '/' || d.path as filepath, @@ -4491,23 +4501,11 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle JOIN content ON content.hash = d.hash LEFT JOIN document_metadata dm ON dm.document_id = d.id WHERE d.active = 1 + ORDER BY fm.bm25_score ASC + LIMIT ? `; - if (collectionFilter) { - sql += ` AND d.collection = ?`; - params.push(String(collectionFilter)); - } - - if (filter) { - // Only documents with current, error-free extraction can match — an - // unprocessed document must not accidentally satisfy `exists: false`. - const compiledFilter = compileMetadataFilter(filter, "d"); - sql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`; - params.push(...compiledFilter.params); - } - // bm25 lower is better; sort ascending. - sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`; params.push(limit); const rows = db.prepare(sql).all(...params) as { filepath: string; display_path: string; title: string; body: string; hash: string; bm25_score: number; metadata_json: string | null }[]; diff --git a/test/store.test.ts b/test/store.test.ts index 92c030d28..0b5c7ddf5 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -4332,6 +4332,44 @@ describe("Vector Search collection filter", () => { await cleanupTestDb(store); }); + + test("searchFTS finds a scoped doc the global candidate window does not contain", async () => { + const store = await createTestStore(); + const large = await createTestCollection({ name: "large-fts", pwd: "/test/large-fts" }); + const small = await createTestCollection({ name: "small-fts", pwd: "/test/small-fts" }); + + // 40 short noise docs carrying the term in the title (bm25 title weight + // 4.0), so every one of them outranks the target globally. + for (let i = 0; i < 40; i++) { + await insertTestDocument(store.db, large, { + name: `noise-${i}`, + title: `zebra zebra ${i}`, + body: `# Noise ${i}\n\nzebra zebra zebra, noise document ${i}.`, + displayPath: `noise-${i}.md`, + }); + } + + // One long target doc that mentions the term once, dead last globally. + await insertTestDocument(store.db, small, { + name: "target", + title: "Target", + body: `# Target\n\n${"filler prose without the search term. ".repeat(40)}zebra.`, + displayPath: "target.md", + }); + + // Unscoped, the target is nowhere near the top of the ranking. + const global = store.searchFTS("zebra", 2); + expect(global).toHaveLength(2); + expect(global.every(r => r.collectionName === large)).toBe(true); + + // Scoped, it is the only answer. Taking a global top-(limit * 10) and + // filtering afterwards returned nothing here: all 20 candidates were + // large-fts documents. + const scoped = store.searchFTS("zebra", 2, small); + expect(scoped.map(r => r.displayPath)).toEqual([`${small}/target.md`]); + + await cleanupTestDb(store); + }); }); // ============================================================================= From 8ead9a082c445506402b6eb61458a6070efe8e6e Mon Sep 17 00:00:00 2001 From: Oliver Ratzesberger Date: Sat, 22 Aug 2026 00:30:49 +0200 Subject: [PATCH 06/82] perf(mcp): cache server instructions instead of rebuilding per session createMcpServer() rebuilds the server instructions on every call, and the Streamable HTTP transport creates one McpServer per MCP session - so every initialize re-runs the full index-status aggregation behind buildInstructions(). On a large index that recomputation dominates the handshake, and agent reconnect bursts stack the status scans (observed as multi-second to tens-of-seconds initializes on an 81k-row index). Cache the built instructions per store with a 60s TTL: concurrent initializes share one in-flight build (a reconnect storm now costs one status scan instead of N), failed builds are dropped so the next initialize retries, and index updates from other processes still surface within a minute without restarting the server. The status tool is untouched and always live. Co-Authored-By: Claude (cherry picked from commit dcb58d3afa3adf7fa009441381e82351be6bfa71) --- src/mcp/server.ts | 30 ++++++++++++++++++++++++++++- test/mcp.test.ts | 49 +++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 78 insertions(+), 1 deletion(-) diff --git a/src/mcp/server.ts b/src/mcp/server.ts index 41b3a0e75..9335b65f5 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -213,6 +213,34 @@ async function buildInstructions(store: QMDStore): Promise { return lines.join("\n"); } +/** + * Cache built instructions per store. The HTTP transport builds a fresh + * McpServer per request (sessionless MCP 2026-07-28), and buildInstructions() + * runs a full index-status aggregation, so rebuilding it per request puts that + * scan in front of every call. Concurrent builds share one in-flight promise, + * and entries expire after a short TTL so index updates from other processes + * still surface without a server restart. + */ +const INSTRUCTIONS_CACHE_TTL_MS = 60_000; +const instructionsCache = new WeakMap; builtAt: number }>(); + +function getInstructions(store: QMDStore): Promise { + const cached = instructionsCache.get(store); + if (cached && Date.now() - cached.builtAt < INSTRUCTIONS_CACHE_TTL_MS) { + return cached.promise; + } + const promise = buildInstructions(store); + instructionsCache.set(store, { promise, builtAt: Date.now() }); + // Drop failed builds so the next initialize retries instead of replaying + // the cached rejection for the rest of the TTL window. + promise.catch(() => { + if (instructionsCache.get(store)?.promise === promise) { + instructionsCache.delete(store); + } + }); + return promise; +} + /** * Create an MCP server with all QMD tools, resources, and prompts registered. * Shared by both stdio and HTTP transports. @@ -224,7 +252,7 @@ async function createMcpServer(store: QMDStore, inflight?: InflightGate): Promis const server = new McpServer( { name: "qmd", version: getPackageVersion() }, { - instructions: await buildInstructions(store), + instructions: await getInstructions(store), // tools/list is static for the process lifetime; resources/read stays // uncacheable because the index can change under us. cacheHints: { diff --git a/test/mcp.test.ts b/test/mcp.test.ts index ecb5f9664..dc75662e9 100644 --- a/test/mcp.test.ts +++ b/test/mcp.test.ts @@ -1124,6 +1124,55 @@ describe.skipIf(!!process.env.CI)("MCP HTTP Transport", () => { expect(json.result.serverInfo.name).toBe("qmd"); }); + test("POST /mcp initialize serves cached instructions across requests", async () => { + // The legacy initialize path answers either JSON or a single SSE frame + // depending on how the transport negotiates; read both the same way. + const legacyInitialize = async (id: number) => { + const res = await fetch(`${baseUrl}/mcp`, { + method: "POST", + headers: { + "Content-Type": "application/json", + "Accept": "application/json, text/event-stream", + }, + body: JSON.stringify({ + jsonrpc: "2.0", + id, + method: "initialize", + params: { + protocolVersion: "2025-03-26", + capabilities: {}, + clientInfo: { name: "test-client", version: "1.0.0" }, + }, + }), + }); + const text = await res.text(); + const data = text.includes("data:") + ? text.split("\n").find(line => line.startsWith("data:"))!.slice(5).trim() + : text; + return JSON.parse(data) as any; + }; + + const first = await legacyInitialize(10); + const firstInstructions = first.result.instructions; + expect(typeof firstInstructions).toBe("string"); + expect(firstInstructions.length).toBeGreaterThan(0); + + // Change the index under the server. HTTP builds a fresh McpServer per + // request, so without the cache the next initialize rebuilds the + // instructions and reports the new document count; with it, the same + // string is served until the TTL lapses. + const db = openDatabase(httpTestDbPath); + const now = new Date().toISOString(); + db.prepare(`INSERT OR IGNORE INTO content (hash, doc, created_at) VALUES (?, ?, ?)`) + .run("hash-instr-cache", "# Cache probe\nbody", now); + db.prepare(`INSERT INTO documents (collection, path, title, hash, created_at, modified_at, active) VALUES ('docs', ?, ?, ?, ?, ?, 1)`) + .run("instructions-cache-probe.md", "Cache Probe", "hash-instr-cache", now, now); + db.close(); + + const second = await legacyInitialize(11); + expect(second.result.instructions).toBe(firstInstructions); + }); + test("POST /mcp tools/list returns registered tools without a session", async () => { const { status, json, contentType, headers } = await mcpRequest("tools/list"); expect(status).toBe(200); From dc4732bc682104de2a470ff6faad6bce44e4065e Mon Sep 17 00:00:00 2001 From: Oliver Ratzesberger Date: Sat, 22 Aug 2026 07:14:51 +0200 Subject: [PATCH 07/82] fix(store): scope FTS by materializing the match set, not a per-rowid prefilter The rowid IN (...) prefilter defeats FTS5 early termination: SQLite drives the query from the collection's whole doclist and probes the FTS5 index once per candidate rowid (EXPLAIN: VIRTUAL TABLE INDEX 0:=M3), about 55ms per in-scope document. On the live index that is 0.02s unscoped versus 21s scoped to a large collection, and 507s for one pass over 19 collections, which on a single-threaded MCP server is a multi-minute outage for every client. Keep the MATCH global and unbounded, MATERIALIZE the complete match set once (corpus-bounded), and apply the collection filter, ORDER BY and LIMIT in the outer query. Correct (the inner set is complete, so the outer filter cannot truncate it) and fast (natural FTS5 operation, EXPLAIN VIRTUAL TABLE INDEX 0:M3). MATERIALIZED is load-bearing: without it the planner may flatten the CTE and refold the collection predicate into the MATCH. Unscoped path unchanged; result rows identical to the prefilter. (cherry picked from commit 0c7a0b8cfa3551dcec3b5fa53baf665c0231bf77) --- src/store.ts | 124 ++++++++++++++++++++++++++++++++------------------- 1 file changed, 77 insertions(+), 47 deletions(-) diff --git a/src/store.ts b/src/store.ts index cb37cf977..6e9f412d5 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4456,58 +4456,88 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle const ftsQuery = buildFTS5Query(query); if (!ftsQuery) return []; - // Use a CTE to keep FTS5 driving the query. Without it, SQLite's planner - // combines the FTS5 MATCH and the collection filter into a single WHERE - // clause, which can make it abandon the FTS5 index and fall back to a full - // scan, turning an 8ms query into a 17-second query on large collections. + // Two SQL shapes, chosen by whether a collection scope is present, because + // the correct-and-fast plan differs: + // + // Unscoped: the FTS5 MATCH is the whole answer, so bound the CTE with the + // requested LIMIT and let FTS5 early-terminate once the top-N are settled. + // + // Scoped: the collection filter must be applied to a COMPLETE match set, + // never to a pre-truncated one. A LIMIT inside the CTE (the earlier + // `limit * 10` over-fetch) is a correctness bug: a collection holding a + // small share of the index loses every row whenever the global top-N holds + // none of its documents, and the caller gets an empty result that reads as + // "this collection has nothing on the subject". But the obvious correct + // form, pushing the collection into the CTE as a + // `rowid IN (SELECT id FROM documents WHERE collection = ?)` prefilter, + // defeats FTS5 early termination: SQLite drives the query from the + // collection's whole doclist and probes the FTS5 index once per candidate + // rowid (EXPLAIN: `... VIRTUAL TABLE INDEX 0:=M3`), which measured about + // 55ms per in-scope document, 0.02s unscoped versus 21s scoped to a large + // collection. + // + // So the scoped shape keeps the MATCH global and unbounded, MATERIALIZEs + // the complete match set once (corpus-bounded, at most one row per matching + // document), and applies the collection filter, ORDER BY and LIMIT in the + // outer query. MATERIALIZED is load-bearing: without it the planner may + // flatten the CTE and fold the collection predicate back into the MATCH, + // reintroducing exactly the per-rowid plan this avoids. + const params: (string | number)[] = [ftsQuery]; - // Scope the FTS lookup itself with a rowid prefilter instead of taking a - // global top-N and dropping the out-of-scope rows afterwards. The over-fetch - // this replaces (limit * 10 candidates when scoped) could not make a - // post-filter correct, it only moved the cutoff: a collection holding a - // small share of the index still lost every row whenever the global top-N - // happened to contain none of its documents, and the caller got an empty - // result that reads as "this collection has nothing on the subject". - // - // documents_fts.rowid is documents.id, so the subquery is a covering-index - // read on idx_documents_collection and the MATCH still runs against the FTS5 - // index (EXPLAIN QUERY PLAN: SCAN documents_fts VIRTUAL TABLE INDEX 0:=M3), - // which is what the CTE above exists to protect. - let scopeFilter = ""; + let sql: string; if (collectionFilter) { - scopeFilter = `\n AND rowid IN (SELECT id FROM documents WHERE active = 1 AND collection = ?)`; - params.push(String(collectionFilter)); + sql = ` + WITH fts_matches AS MATERIALIZED ( + SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score + FROM documents_fts + WHERE documents_fts MATCH ? + ) + SELECT + 'qmd://' || d.collection || '/' || d.path as filepath, + d.collection || '/' || d.path as display_path, + d.title, + ${cappedBodySql("content.doc")} as body, + d.hash, + fm.bm25_score, + dm.metadata_json + FROM fts_matches fm + JOIN documents d ON d.id = fm.rowid + JOIN content ON content.hash = d.hash + LEFT JOIN document_metadata dm ON dm.document_id = d.id + WHERE d.active = 1 AND d.collection = ? + ORDER BY fm.bm25_score ASC + LIMIT ? + `; + params.push(String(collectionFilter), limit); + } else { + sql = ` + WITH fts_matches AS ( + SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score + FROM documents_fts + WHERE documents_fts MATCH ? + ORDER BY bm25_score ASC + LIMIT ${limit} + ) + SELECT + 'qmd://' || d.collection || '/' || d.path as filepath, + d.collection || '/' || d.path as display_path, + d.title, + ${cappedBodySql("content.doc")} as body, + d.hash, + fm.bm25_score, + dm.metadata_json + FROM fts_matches fm + JOIN documents d ON d.id = fm.rowid + JOIN content ON content.hash = d.hash + LEFT JOIN document_metadata dm ON dm.document_id = d.id + WHERE d.active = 1 + ORDER BY fm.bm25_score ASC + LIMIT ? + `; + params.push(limit); } - const sql = ` - WITH fts_matches AS ( - SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score - FROM documents_fts - WHERE documents_fts MATCH ?${scopeFilter} - ORDER BY bm25_score ASC - LIMIT ${limit} - ) - SELECT - 'qmd://' || d.collection || '/' || d.path as filepath, - d.collection || '/' || d.path as display_path, - d.title, - ${cappedBodySql("content.doc")} as body, - d.hash, - fm.bm25_score, - dm.metadata_json - FROM fts_matches fm - JOIN documents d ON d.id = fm.rowid - JOIN content ON content.hash = d.hash - LEFT JOIN document_metadata dm ON dm.document_id = d.id - WHERE d.active = 1 - ORDER BY fm.bm25_score ASC - LIMIT ? - `; - - // bm25 lower is better; sort ascending. - params.push(limit); - const rows = db.prepare(sql).all(...params) as { filepath: string; display_path: string; title: string; body: string; hash: string; bm25_score: number; metadata_json: string | null }[]; return rows.map(row => { const collectionName = row.filepath.split('//')[1]?.split('/')[0] || ""; From 9e1e8a5507dcfa8073fc11b483406516a54642a0 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:37:53 -0500 Subject: [PATCH 08/82] test(mcp): cover the per-store server instructions cache #815 caches the MCP server instructions per store for 60 seconds, but its own test sits in a describe that CI skips. These tests run in the CI-enabled 2026-07-28 protocol describe through server/discover: - a document added between two discovers leaves the instructions unchanged; - once the clock moves past 60 seconds, the next discover rebuilds and reports the live document count; - a build that fails is dropped rather than cached, so the next discover rebuilds, and that result is then cached. --- test/mcp.test.ts | 88 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 87 insertions(+), 1 deletion(-) diff --git a/test/mcp.test.ts b/test/mcp.test.ts index dc75662e9..70ea56a2c 100644 --- a/test/mcp.test.ts +++ b/test/mcp.test.ts @@ -5,7 +5,7 @@ * Uses mocked Ollama responses and a test database. */ -import { describe, test, expect, beforeAll, afterAll, beforeEach, afterEach } from "vitest"; +import { describe, test, expect, vi, beforeAll, afterAll, beforeEach, afterEach } from "vitest"; import { openDatabase, loadSqliteVec } from "../src/db.js"; import type { Database } from "../src/db.js"; import { getDefaultLlamaCpp, disposeDefaultLlamaCpp } from "../src/llm"; @@ -1429,6 +1429,92 @@ describe("MCP HTTP Transport — 2026-07-28 protocol", () => { expect(res.status).toBe(405); expect(res.headers.get("mcp-session-id")).toBeNull(); }); + + // Each request gets a fresh McpServer, so these observe the per-store + // instructions cache through server/discover. + async function discoverInstructions(id: number): Promise<{ status: number; instructions?: string }> { + const { status, json } = await postMcp( + { jsonrpc: "2.0", id, method: "server/discover", params: { _meta: mcp2026Meta } }, + { "MCP-Protocol-Version": MCP_2026, "Mcp-Method": "server/discover" }, + ); + return { status, instructions: json.result?.instructions }; + } + + function documentCount(instructions: string | undefined): number { + const match = instructions?.match(/over (\d+) markdown documents/); + if (!match) throw new Error(`no document count in instructions: ${instructions}`); + return Number(match[1]); + } + + function activeDocuments(): number { + const db = openDatabase(dbPath); + try { + const row = db.prepare(`SELECT COUNT(*) AS n FROM documents WHERE active = 1`).get() as { n: number }; + return row.n; + } finally { + db.close(); + } + } + + function addDocument(slug: string): void { + const db = openDatabase(dbPath); + const now = new Date().toISOString(); + db.prepare(`INSERT INTO content (hash, doc, created_at) VALUES (?, ?, ?)`) + .run(`hash-${slug}`, `# ${slug}\nbody`, now); + db.prepare(`INSERT INTO documents (collection, path, title, hash, created_at, modified_at, active) VALUES ('docs', ?, ?, ?, ?, ?, 1)`) + .run(`${slug}.md`, slug, `hash-${slug}`, now, now); + db.close(); + } + + test("server/discover serves cached instructions after the index changes", async () => { + const first = await discoverInstructions(20); + expect(first.status).toBe(200); + addDocument("instructions-cache-hit"); + const second = await discoverInstructions(21); + expect(second.status).toBe(200); + expect(second.instructions).toBe(first.instructions); + }); + + test("server/discover rebuilds instructions once the 60 s cache entry expires", async () => { + const realNow = Date.now.bind(Date); + const first = await discoverInstructions(22); + addDocument("instructions-cache-expiry"); + const cached = await discoverInstructions(23); + expect(cached.instructions).toBe(first.instructions); + const clock = vi.spyOn(Date, "now").mockImplementation(() => realNow() + 61_000); + try { + const rebuilt = await discoverInstructions(24); + expect(rebuilt.status).toBe(200); + expect(documentCount(rebuilt.instructions)).toBe(activeDocuments()); + expect(documentCount(rebuilt.instructions)).toBeGreaterThan(documentCount(first.instructions)); + } finally { + clock.mockRestore(); + } + }); + + test("server/discover retries a failed instructions build instead of caching the failure", async () => { + const realNow = Date.now.bind(Date); + // Past every earlier entry's TTL, so the next discover has to build. + const clock = vi.spyOn(Date, "now").mockImplementation(() => realNow() + 10 * 60_000); + const db = openDatabase(dbPath); + try { + db.exec(`ALTER TABLE store_collections RENAME TO store_collections_offline`); + const failed = await discoverInstructions(25); + expect(failed.instructions).toBeUndefined(); + db.exec(`ALTER TABLE store_collections_offline RENAME TO store_collections`); + + const retried = await discoverInstructions(26); + expect(retried.status).toBe(200); + addDocument("instructions-cache-after-failure"); + const cached = await discoverInstructions(27); + expect(cached.instructions).toBe(retried.instructions); + } finally { + const offline = db.prepare(`SELECT name FROM sqlite_master WHERE name = 'store_collections_offline'`).get(); + if (offline) db.exec(`ALTER TABLE store_collections_offline RENAME TO store_collections`); + db.close(); + clock.mockRestore(); + } + }); }); From 6589100ea4c687aa615ac012cfee2aee179bbb39 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:04:40 -0500 Subject: [PATCH 09/82] fix(mcp): rebuild cached instructions stamped ahead of the clock #815's cache treats an entry as fresh while now minus its build time is under 60 seconds, which a negative age also satisfies. After the wall clock steps back, an entry built before the step stays cached for the size of the step plus a minute, past the bound the cache promises. The tests that move Date.now forward left such entries behind as well. An entry with a negative age now counts as stale and is rebuilt on the next request. --- src/mcp/server.ts | 5 ++++- test/mcp.test.ts | 14 ++++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/src/mcp/server.ts b/src/mcp/server.ts index 9335b65f5..502b2c797 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -226,7 +226,10 @@ const instructionsCache = new WeakMap; buil function getInstructions(store: QMDStore): Promise { const cached = instructionsCache.get(store); - if (cached && Date.now() - cached.builtAt < INSTRUCTIONS_CACHE_TTL_MS) { + // A negative age means the clock stepped back since the build; that entry is + // not known to be fresh, so it is rebuilt rather than kept past the TTL. + const age = cached ? Date.now() - cached.builtAt : -1; + if (cached && age >= 0 && age < INSTRUCTIONS_CACHE_TTL_MS) { return cached.promise; } const promise = buildInstructions(store); diff --git a/test/mcp.test.ts b/test/mcp.test.ts index 70ea56a2c..2a17ba84e 100644 --- a/test/mcp.test.ts +++ b/test/mcp.test.ts @@ -1515,6 +1515,20 @@ describe("MCP HTTP Transport — 2026-07-28 protocol", () => { clock.mockRestore(); } }); + + test("server/discover rebuilds an entry stamped ahead of the clock once the clock steps back", async () => { + const realNow = Date.now.bind(Date); + // Built while the clock read ten minutes ahead, as after a backward step. + const clock = vi.spyOn(Date, "now").mockImplementation(() => realNow() + 20 * 60_000); + const ahead = await discoverInstructions(28); + clock.mockRestore(); + addDocument("instructions-cache-clock-step"); + + const rebuilt = await discoverInstructions(29); + expect(rebuilt.status).toBe(200); + expect(rebuilt.instructions).not.toBe(ahead.instructions); + expect(documentCount(rebuilt.instructions)).toBe(activeDocuments()); + }); }); From 0d0c3504ee66e8c47c37b4e57f4c212b05be2c17 Mon Sep 17 00:00:00 2001 From: ilepn <26382655+ilepn@users.noreply.github.com> Date: Mon, 24 Aug 2026 13:31:20 +0300 Subject: [PATCH 10/82] perf: avoid redundant CLI runtime spawn (cherry picked from commit dbc829ad22e212c064fd3d87566e2b3ee17c647b) --- CHANGELOG.md | 5 +++ bin/qmd | 76 ++++++++++++++++++++-------------- scripts/package-smoke.mjs | 8 ++++ test/bin-wrapper.test.ts | 85 ++++++++++++++++++++++++++++++++------- 4 files changed, 129 insertions(+), 45 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index bb094c48e..ae5792140 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,11 @@ keeps chunk boundaries aligned with the model that creates and verifies the stored vectors without initializing an unrelated provider. +### Changed + +- Avoid a redundant runtime process on packaged CLI calls where the launcher is + already running under its selected Node or Bun runtime. + ## [2.8.3] - 2026-08-16 ### Security diff --git a/bin/qmd b/bin/qmd index 3ae0309d6..981663ab6 100755 --- a/bin/qmd +++ b/bin/qmd @@ -11,7 +11,7 @@ import { spawn, spawnSync } from "node:child_process"; import { existsSync, realpathSync } from "node:fs"; import { dirname, resolve, sep } from "node:path"; -import { fileURLToPath } from "node:url"; +import { fileURLToPath, pathToFileURL } from "node:url"; // Resolve symlinks so global installs (npm link / npm install -g) can find // the actual package directory instead of the global bin directory. @@ -160,33 +160,47 @@ if (existsSync(resolve(pkgDir, "package-lock.json"))) { } const selected = useSourceMode ? sourceRunner : (runnerName === "node" ? "node" : "bun"); -// Pin Node to the binary that launched this trampoline. Native addons -// (better-sqlite3) are compiled for that install's ABI; spawning PATH -// `node` picks up nvm/fnm/mise shims in the project directory and fails -// with NODE_MODULE_VERSION / ERR_DLOPEN_FAILED (leftover from #577; #319). -// If the trampoline itself is running under bun (`bun bin/qmd`), keep -// PATH `node` so we don't load Node-ABI addons into bun. -const runner = selected === "node" && typeof process.versions.bun !== "string" - ? process.execPath - : selected; -const args = useSourceMode ? sourceArgs : [jsEntry, ...process.argv.slice(2)]; -const needsShell = (runner === "bun") && process.platform === "win32"; - -const child = spawn(runner, args, { - stdio: "inherit", - shell: needsShell, -}); - -child.on("exit", (code, signal) => { - if (signal) { - process.kill(process.pid, signal); - } else { - process.exit(code ?? 0); - } -}); - -child.on("error", (err) => { - const name = useSourceMode ? sourceRunner : runnerName; - console.error(`qmd: failed to launch ${name}: ${err.message}`); - process.exit(1); -}); +const launchedByBun = typeof process.versions.bun === "string"; +const runtimeAlreadySelected = !useSourceMode + && ((selected === "node" && !launchedByBun) || (selected === "bun" && launchedByBun)); + +// Published packages can reach this launcher under the runtime selected by +// their lockfile. Starting that same runtime again adds an avoidable process +// startup to matching-runtime CLI calls. Align argv with the compiled entry +// point and import it in place; the CLI's main-module guard then behaves +// exactly as in a direct run. +if (runtimeAlreadySelected) { + process.argv.splice(1, 1, jsEntry); + await import(pathToFileURL(jsEntry).href); +} else { + const args = useSourceMode ? sourceArgs : [jsEntry, ...process.argv.slice(2)]; + // Pin Node to the binary that launched this trampoline. Native addons + // (better-sqlite3) are compiled for that install's ABI; spawning PATH + // `node` picks up nvm/fnm/mise shims in the project directory and fails + // with NODE_MODULE_VERSION / ERR_DLOPEN_FAILED (leftover from #577; #319). + // If the trampoline itself is running under bun (`bun bin/qmd`), keep + // PATH `node` so we don't load Node-ABI addons into bun. + const runner = selected === "node" && !launchedByBun + ? process.execPath + : selected; + const needsShell = (runner === "bun") && process.platform === "win32"; + + const child = spawn(runner, args, { + stdio: "inherit", + shell: needsShell, + }); + + child.on("exit", (code, signal) => { + if (signal) { + process.kill(process.pid, signal); + } else { + process.exit(code ?? 0); + } + }); + + child.on("error", (err) => { + const name = useSourceMode ? sourceRunner : runnerName; + console.error(`qmd: failed to launch ${name}: ${err.message}`); + process.exit(1); + }); +} diff --git a/scripts/package-smoke.mjs b/scripts/package-smoke.mjs index f9622e7a4..2a7ddae8f 100644 --- a/scripts/package-smoke.mjs +++ b/scripts/package-smoke.mjs @@ -57,9 +57,17 @@ assertPath("dist/cli/qmd.js", "compiled CLI"); run("compiled CLI under Node", process.execPath, ["dist/cli/qmd.js", "--help"], { quiet: true }); run("package wrapper", "sh", ["bin/qmd", "--help"], { quiet: true }); +run("dist package wrapper under Node", process.execPath, ["bin/qmd", "--help"], { + quiet: true, + env: { ...process.env, QMD_SOURCE_MODE: "0" }, +}); if (process.env.QMD_SKIP_BUN_SMOKE === "1") { console.log("==> compiled CLI under Bun (skipped by QMD_SKIP_BUN_SMOKE=1)"); } else { run("compiled CLI under Bun", "bun", ["dist/cli/qmd.js", "--help"], { quiet: true }); + run("dist package wrapper under Bun", "bun", ["bin/qmd", "--help"], { + quiet: true, + env: { ...process.env, QMD_SOURCE_MODE: "0" }, + }); } diff --git a/test/bin-wrapper.test.ts b/test/bin-wrapper.test.ts index ae31bb19a..f0592a82f 100644 --- a/test/bin-wrapper.test.ts +++ b/test/bin-wrapper.test.ts @@ -1,7 +1,7 @@ import { afterEach, describe, expect, test } from "vitest"; import { chmodSync, copyFileSync, mkdtempSync, mkdirSync, readFileSync, realpathSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; -import { dirname, join, relative } from "node:path"; +import { delimiter, dirname, join, relative } from "node:path"; import { execFileSync, spawnSync } from "node:child_process"; import { fileURLToPath } from "node:url"; @@ -58,24 +58,35 @@ fi return { root, capturePath, runtimeBin }; } +function guardedEsmCli(body: string): string { + return [ + 'import { realpathSync, writeFileSync } from "node:fs";', + 'import { fileURLToPath } from "node:url";', + 'const __filename = fileURLToPath(import.meta.url);', + 'const argv1 = process.argv[1];', + 'const isMain = argv1 === __filename', + ' || argv1?.endsWith("/qmd.ts")', + ' || argv1?.endsWith("/qmd.js")', + ' || (argv1 != null && realpathSync(argv1) === __filename);', + 'if (isMain) {', + ` ${body}`, + '}', + '', + ].join("\n"); +} + function makePackage(root: string, packagePath: string, lockfiles: string[] = [], options: { dist?: boolean; source?: boolean; tsx?: boolean; git?: boolean } = {}) { const packageRoot = join(root, packagePath); const includeDist = options.dist ?? true; mkdirSync(join(packageRoot, "bin"), { recursive: true }); + writeFileSync(join(packageRoot, "package.json"), '{"type":"module"}\n'); copyFileSync(join(repoRoot, "bin", "qmd"), join(packageRoot, "bin", "qmd")); chmodSync(join(packageRoot, "bin", "qmd"), 0o755); if (includeDist) { mkdirSync(join(packageRoot, "dist", "cli"), { recursive: true }); writeFileSync( join(packageRoot, "dist", "cli", "qmd.js"), - [ - 'const { writeFileSync } = require("node:fs");', - 'const capture = process.env.QMD_WRAPPER_CAPTURE;', - 'if (capture) {', - ' writeFileSync(capture, ["node", process.argv[1], ...process.argv.slice(2)].join("\\n") + "\\n");', - '}', - '', - ].join("\n"), + guardedEsmCli('writeFileSync(process.env.QMD_WRAPPER_CAPTURE, ["node", process.argv[1], ...process.argv.slice(2)].join("\\n") + "\\n");'), ); } if (options.source) { @@ -299,15 +310,61 @@ describe("bin/qmd package wrapper", () => { expect(result.args).toEqual([realpathSync(join(packageRoot, "src", "cli", "qmd.ts")), "--version"]); }); - test("node child uses process.execPath, not a different PATH node (#577 leftover)", () => { + test("source-mode Node child uses process.execPath, not a different PATH node (#577 leftover)", () => { const { root, runtimeBin, capturePath } = makeTempFixture(); + const packageRoot = makePackage(root, "qmd", [], { source: true, tsx: true, git: true }); + + const child = spawnSync(REAL_NODE, [join(packageRoot, "bin", "qmd"), "--version"], { + env: { + ...process.env, + PATH: `${runtimeBin}${delimiter}${process.env.PATH ?? ""}`, + QMD_WRAPPER_CAPTURE: capturePath, + }, + encoding: "utf8", + }); + const [runtime, scriptPath, ...args] = readFileSync(capturePath, "utf8").trimEnd().split("\n"); + + expect(child.error).toBeUndefined(); + expect(child.status).toBe(0); + expect(runtime).toBe("node"); + expect(scriptPath).toBe(realpathSync(join(packageRoot, "node_modules", "tsx", "dist", "cli.mjs"))); + expect(args).toEqual([realpathSync(join(packageRoot, "src", "cli", "qmd.ts")), "--version"]); + }); + + test("imports dist in the current Node process when Node is already selected", () => { + const { root, capturePath } = makeTempFixture(); const packageRoot = makePackage(root, "node_modules/@tobilu/qmd"); + writeFileSync( + join(packageRoot, "dist", "cli", "qmd.js"), + guardedEsmCli('writeFileSync(process.env.QMD_WRAPPER_CAPTURE, String(process.pid));'), + ); - const result = runWrapper(join(packageRoot, "bin", "qmd"), runtimeBin, capturePath); + const result = spawnSync(REAL_NODE, [join(packageRoot, "bin", "qmd"), "--version"], { + env: { ...process.env, QMD_WRAPPER_CAPTURE: capturePath }, + encoding: "utf8", + }); - expect(result.runtime).toBe("node"); - expect(result.scriptPath).toBe(realpathSync(join(packageRoot, "dist", "cli", "qmd.js"))); - expect(result.args).toEqual(["--version"]); + expect(result.error).toBeUndefined(); + expect(result.status).toBe(0); + expect(Number(readFileSync(capturePath, "utf8"))).toBe(result.pid); + }); + + test.skipIf(typeof process.versions.bun !== "string")("imports dist in the current Bun process when Bun is already selected", () => { + const { root, capturePath } = makeTempFixture(); + const packageRoot = makePackage(root, "node_modules/@tobilu/qmd", ["bun.lock"]); + writeFileSync( + join(packageRoot, "dist", "cli", "qmd.js"), + guardedEsmCli('writeFileSync(process.env.QMD_WRAPPER_CAPTURE, String(process.pid));'), + ); + + const result = spawnSync(process.execPath, [join(packageRoot, "bin", "qmd"), "--version"], { + env: { ...process.env, QMD_WRAPPER_CAPTURE: capturePath }, + encoding: "utf8", + }); + + expect(result.error).toBeUndefined(); + expect(result.status).toBe(0); + expect(Number(readFileSync(capturePath, "utf8"))).toBe(result.pid); }); test("explains how to build when dist is missing and source cannot run", () => { From 150b868be54489261a44ff36b655112f54bca114 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 10:57:23 -0500 Subject: [PATCH 11/82] test(store): cover #918's scoped keyword search across several collections #918's test scopes to one collection. With a scope of several, the search still ran once per collection, and each of those runs filtered a global top (limit * 10) window, so a large collection could crowd out every small one. These build 40 (then 250) short noise documents that outrank a long target in each of two small collections, and expect both targets: - through searchFTS with limit 2; - through structuredSearch, which asks for 20 and so needed more than 200 noise documents to empty the old window. --- test/store.test.ts | 53 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/test/store.test.ts b/test/store.test.ts index 0b5c7ddf5..e7717e01f 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3605,6 +3605,59 @@ describe("Reindex Collection file sync state (#962)", () => { }); }); +describe("Collection-scoped keyword search", () => { + // Short noise documents with the term in their titles outrank, globally, one + // long target in each of two small collections. The old scoped search took a + // global top (limit * 10) and filtered it, so the noise must fill that window. + async function crowdedCollections(store: Store, noiseCount: number): Promise<{ smallA: string; smallB: string }> { + const large = await createTestCollection({ name: "crowd-large", pwd: "/test/crowd-large" }); + const smallA = await createTestCollection({ name: "crowd-small-a", pwd: "/test/crowd-small-a" }); + const smallB = await createTestCollection({ name: "crowd-small-b", pwd: "/test/crowd-small-b" }); + for (let i = 0; i < noiseCount; i++) { + await insertTestDocument(store.db, large, { + name: `noise-${i}`, + title: `zebra zebra ${i}`, + body: `# Noise ${i}\n\nzebra zebra zebra, noise document ${i}.`, + displayPath: `noise-${i}.md`, + }); + } + for (const collection of [smallA, smallB]) { + await insertTestDocument(store.db, collection, { + name: `target-${collection}`, + title: "Target", + body: `# Target\n\n${"filler prose without the search term. ".repeat(40)}zebra.`, + displayPath: "target.md", + }); + } + return { smallA, smallB }; + } + + test("searchFTS over two crowded-out collections returns a match from each", async () => { + const store = await createTestStore(); + try { + const { smallA, smallB } = await crowdedCollections(store, 40); + const results = store.searchFTS("zebra", 2, [smallA, smallB]); + expect(results.map(r => r.displayPath).sort()).toEqual([`${smallA}/target.md`, `${smallB}/target.md`]); + } finally { + await cleanupTestDb(store); + } + }); + + test("structuredSearch over two crowded-out collections returns a match from each", async () => { + const store = await createTestStore(); + try { + // structuredSearch asks searchFTS for 20 results, a window of 200. + const { smallA, smallB } = await crowdedCollections(store, 250); + const results = await structuredSearch(store, [{ type: "lex", query: "zebra" }], { + collections: [smallA, smallB], skipRerank: true, + }); + expect(results.map(r => r.displayPath).sort()).toEqual([`${smallA}/target.md`, `${smallB}/target.md`]); + } finally { + await cleanupTestDb(store); + } + }); +}); + // ============================================================================= // Index Status Tests // ============================================================================= From f4633f92a99ffdb2d0e885ed391fd8169133d70e Mon Sep 17 00:00:00 2001 From: Rohit <83605437+Mr-Beasley@users.noreply.github.com> Date: Mon, 14 Sep 2026 12:01:08 +0000 Subject: [PATCH 12/82] fix(store): apply BM25 scope to the full match set, not a top-N window searchFTS took a global top `limit * 10` from FTS5 and only then applied the collection or metadata filter. When stronger out-of-scope matches filled that window, scoped search returned [] although the scope held matches (#922). Multi-collection scopes inherited it through the #775 per-collection fan-out. Scoped queries now MATERIALIZE the complete MATCH set and filter, order and limit in the outer query. MATERIALIZED stops the planner flattening the CTE into a per-rowid probe. Unscoped queries keep the inner LIMIT. Approach from #918, carried onto the current store (metadata filter). Fixes #922 Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit 1a554c75771ea34d2897eeb8eac830b836227723) --- CHANGELOG.md | 8 +++ src/store.ts | 127 ++++++++++++++--------------------- test/metadata-search.test.ts | 18 +++++ test/store.test.ts | 40 +++++++++++ 4 files changed, 116 insertions(+), 77 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f819cc4d8..e9806e5f0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,14 @@ ### Fixed - Filtered vector search binds candidate IDs as one JSON list, so a parser-valid metadata filter cannot exhaust Node's SQL variable limit during document lookup. Applies to both exact scans and the capped global fallback. +- Collection- and metadata-scoped BM25 search (`qmd search -c`, the lex leg of + `qmd query`, MCP, SDK `searchLex`) no longer returns false-empty or + incomplete results when stronger matches outside the scope fill the old + `limit * 10` candidate window (#922). The scope is now applied to the full + FTS5 match set, materialized once, so `search -c ` is exact for + common terms; unscoped search keeps its early-terminating plan. Builds on + the approach in #918. + - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This keeps chunk boundaries aligned with the model that creates and verifies the diff --git a/src/store.ts b/src/store.ts index 6e9f412d5..0993191fb 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4456,88 +4456,61 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle const ftsQuery = buildFTS5Query(query); if (!ftsQuery) return []; - // Two SQL shapes, chosen by whether a collection scope is present, because - // the correct-and-fast plan differs: - // - // Unscoped: the FTS5 MATCH is the whole answer, so bound the CTE with the - // requested LIMIT and let FTS5 early-terminate once the top-N are settled. - // - // Scoped: the collection filter must be applied to a COMPLETE match set, - // never to a pre-truncated one. A LIMIT inside the CTE (the earlier - // `limit * 10` over-fetch) is a correctness bug: a collection holding a - // small share of the index loses every row whenever the global top-N holds - // none of its documents, and the caller gets an empty result that reads as - // "this collection has nothing on the subject". But the obvious correct - // form, pushing the collection into the CTE as a - // `rowid IN (SELECT id FROM documents WHERE collection = ?)` prefilter, - // defeats FTS5 early termination: SQLite drives the query from the - // collection's whole doclist and probes the FTS5 index once per candidate - // rowid (EXPLAIN: `... VIRTUAL TABLE INDEX 0:=M3`), which measured about - // 55ms per in-scope document, 0.02s unscoped versus 21s scoped to a large - // collection. - // - // So the scoped shape keeps the MATCH global and unbounded, MATERIALIZEs - // the complete match set once (corpus-bounded, at most one row per matching - // document), and applies the collection filter, ORDER BY and LIMIT in the - // outer query. MATERIALIZED is load-bearing: without it the planner may - // flatten the CTE and fold the collection predicate back into the MATCH, - // reintroducing exactly the per-rowid plan this avoids. - + // Use a CTE to force FTS5 to run first, then filter by collection. + // Without the CTE, SQLite's query planner combines FTS5 MATCH with the + // collection filter in a single WHERE clause, which can cause it to + // abandon the FTS5 index and fall back to a full scan — turning an 8ms + // query into a 17-second query on large collections. const params: (string | number)[] = [ftsQuery]; - let sql: string; + // Unscoped, the MATCH is the whole answer: LIMIT inside the CTE lets FTS5 + // stop at the requested count. Scoped by collection or metadata, the filter + // must see the COMPLETE match set — any inner LIMIT (the old `limit * 10`) + // returns false-empty results whenever stronger out-of-scope matches fill + // the window (#922). MATERIALIZED keeps the planner from flattening the CTE + // and folding the filter back into the MATCH. The set is corpus-bounded: + // at most one row per matching document. + const scoped = Boolean(collectionFilter || filter); + + let sql = ` + WITH fts_matches AS ${scoped ? "MATERIALIZED " : ""}( + SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score + FROM documents_fts + WHERE documents_fts MATCH ? + ${scoped ? "" : `ORDER BY bm25_score ASC LIMIT ${limit}`} + ) + SELECT + 'qmd://' || d.collection || '/' || d.path as filepath, + d.collection || '/' || d.path as display_path, + d.title, + ${cappedBodySql("content.doc")} as body, + d.hash, + fm.bm25_score, + dm.metadata_json + FROM fts_matches fm + JOIN documents d ON d.id = fm.rowid + JOIN content ON content.hash = d.hash + LEFT JOIN document_metadata dm ON dm.document_id = d.id + WHERE d.active = 1 + `; + if (collectionFilter) { - sql = ` - WITH fts_matches AS MATERIALIZED ( - SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score - FROM documents_fts - WHERE documents_fts MATCH ? - ) - SELECT - 'qmd://' || d.collection || '/' || d.path as filepath, - d.collection || '/' || d.path as display_path, - d.title, - ${cappedBodySql("content.doc")} as body, - d.hash, - fm.bm25_score, - dm.metadata_json - FROM fts_matches fm - JOIN documents d ON d.id = fm.rowid - JOIN content ON content.hash = d.hash - LEFT JOIN document_metadata dm ON dm.document_id = d.id - WHERE d.active = 1 AND d.collection = ? - ORDER BY fm.bm25_score ASC - LIMIT ? - `; - params.push(String(collectionFilter), limit); - } else { - sql = ` - WITH fts_matches AS ( - SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score - FROM documents_fts - WHERE documents_fts MATCH ? - ORDER BY bm25_score ASC - LIMIT ${limit} - ) - SELECT - 'qmd://' || d.collection || '/' || d.path as filepath, - d.collection || '/' || d.path as display_path, - d.title, - ${cappedBodySql("content.doc")} as body, - d.hash, - fm.bm25_score, - dm.metadata_json - FROM fts_matches fm - JOIN documents d ON d.id = fm.rowid - JOIN content ON content.hash = d.hash - LEFT JOIN document_metadata dm ON dm.document_id = d.id - WHERE d.active = 1 - ORDER BY fm.bm25_score ASC - LIMIT ? - `; - params.push(limit); + sql += ` AND d.collection = ?`; + params.push(String(collectionFilter)); + } + + if (filter) { + // Only documents with current, error-free extraction can match — an + // unprocessed document must not accidentally satisfy `exists: false`. + const compiledFilter = compileMetadataFilter(filter, "d"); + sql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`; + params.push(...compiledFilter.params); } + // bm25 lower is better; sort ascending. + sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`; + params.push(limit); + const rows = db.prepare(sql).all(...params) as { filepath: string; display_path: string; title: string; body: string; hash: string; bm25_score: number; metadata_json: string | null }[]; return rows.map(row => { const collectionName = row.filepath.split('//')[1]?.split('/')[0] || ""; diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index 3dbfda44c..8b8c3f864 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -77,6 +77,24 @@ describe("searchFTS with metadata filter", () => { expect(filtered[0]!.metadata).toEqual({ status: "published" }); }); + test("finds a matching document ranked below the unfiltered candidate window", async () => { + // limit=5 made the old filtered window 50; every noise doc outranks the target. + for (let i = 0; i < 50; i++) { + await insertDoc("notes", `noise-${i}.md`, `# N${i}\n\nalpha alpha alpha`, { status: "draft" }); + } + await insertDoc( + "notes", + "target.md", + `# Target\n\n${"Unrelated prose. ".repeat(40)}One weaker mention of alpha.`, + { status: "published" }, + ); + + const filtered = searchFTS(store.db, "alpha", 5, undefined, { + key: "status", operator: "eq", value: "published", + }); + expect(filtered.map(r => r.displayPath)).toEqual(["notes/target.md"]); + }); + test("excludes pending, stale, and errored documents from filtered search", async () => { const { documentId: erroredId } = await insertDoc("notes", "errored.md", "# One\n\ncommon term"); await insertDoc("notes", "pending.md", "# Two\n\ncommon term"); diff --git a/test/store.test.ts b/test/store.test.ts index e7717e01f..7c83d58ec 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -1754,6 +1754,46 @@ describe("FTS Search", () => { await cleanupTestDb(store); }); + test("searchFTS scope is exact when an out-of-scope collection fills the candidate window (#922)", async () => { + const store = await createTestStore(); + const noise = await createTestCollection({ name: "noise", pwd: "/test/noise" }); + const target = await createTestCollection({ name: "target", pwd: "/test/target" }); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + + // limit=5 made the old scoped window limit*10 = 50: exactly as many + // stronger out-of-scope hits as fit in it. + for (let i = 0; i < 50; i++) { + await insertTestDocument(store.db, noise, { + name: `noise-${i}`, + title: "alpha alpha", + body: `Noise ${i}: alpha alpha alpha.`, + displayPath: `noise-${i}.md`, + }); + } + await insertTestDocument(store.db, target, { + name: "t", + title: "Target", + body: `${"Unrelated prose. ".repeat(40)}One weaker mention of alpha.`, + displayPath: "t.md", + }); + await insertTestDocument(store.db, other, { + name: "o", + title: "Other", + body: "Nothing relevant here.", + displayPath: "o.md", + }); + + expect(store.searchFTS("alpha", 5).every(r => r.collectionName === noise)).toBe(true); + + const single = store.searchFTS("alpha", 5, target); + expect(single.map(r => r.displayPath)).toEqual([`${target}/t.md`]); + + const multi = store.searchFTS("alpha", 5, [target, other]); + expect(multi.map(r => r.displayPath)).toEqual([`${target}/t.md`]); + + await cleanupTestDb(store); + }); + test("searchFTS finds CJK documents by exact and mixed queries", async () => { const store = await createTestStore(); const collectionName = await createTestCollection(); From be03377b97fccc286d89166ce38b9cbc0af1c0d5 Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Mon, 14 Sep 2026 18:20:12 -0700 Subject: [PATCH 13/82] refactor(metadata): name the filter condition's target `field` A filter condition is a predicate over one field of the record under evaluation: `field` names it, `operator` says how to compare, `value` is the operand. For a document, the fields are its metadata keys, and that special case is what the property was named after. Rename the property from `key` to `field` so the grammar describes itself without reference to what it happens to be applied to. Nothing else in the AST changes: `operator`, `value`, `operands`, and `operand` keep their names, and the parser rejects the old property the same way it rejects any unknown one, with the JSON path of the failing node. The rename lands before metadata filtering (#910) ships, so no released version accepts `key`. Updates the README filter table and semantics, the skill, the MCP `query` tool description, the CLI help and error examples, the changelog entry, and every test that builds a condition. Assisted-by: Claude Fable 5.1 via Pi --- CHANGELOG.md | 2 +- README.md | 24 +++---- skills/qmd/SKILL.md | 6 +- src/cli/qmd.ts | 4 +- src/mcp/server.ts | 6 +- src/metadata-filter.ts | 48 +++++++------ test/metadata-cli.test.ts | 10 +-- test/metadata-filter.test.ts | 119 +++++++++++++++++---------------- test/metadata-search.test.ts | 24 +++---- test/metadata-surfaces.test.ts | 14 ++-- 10 files changed, 131 insertions(+), 126 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2cb32c5f2..1ad02c921 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,7 +5,7 @@ ### Added - Added Oxlint lint fence. -- Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter `, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator`: `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, and `exists` presence. Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes). +- Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter `, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator` (conditions are `{ field, operator, value }`, where `field` names the metadata key): `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, and `exists` presence. Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes). ### Fixed diff --git a/README.md b/README.md index 76b822b20..40ff97b10 100644 --- a/README.md +++ b/README.md @@ -302,8 +302,8 @@ const published = await store.search({ filter: { operator: "and", operands: [ - { key: "topics", operator: "all", value: ["typescript"] }, - { key: "status", operator: "ne", value: "draft" }, + { field: "topics", operator: "all", value: ["typescript"] }, + { field: "status", operator: "ne", value: "draft" }, ], }, }) @@ -940,24 +940,24 @@ qmd: Supported values are strings, numbers, booleans, and flat homogeneous arrays of one of those. Nested objects, nulls, empty arrays, and mixed-type arrays are rejected (the document still indexes; it is excluded from filtered search until corrected). Metadata keys are user-defined data — `tags`, `topics`, and `labels` are all ordinary keys with no special semantics. -Every search surface (CLI, SDK, MCP, HTTP) accepts the same recursive filter, a JSON AST discriminated by `operator`: +Every search surface (CLI, SDK, MCP, HTTP) accepts the same recursive filter, a JSON AST discriminated by `operator`. A condition is a predicate over one field of the document's metadata: `field` names the metadata key, `operator` says how to compare, and `value` is what to compare against: ```sh # One condition qmd search "authentication" \ - --filter '{"key":"status","operator":"eq","value":"published"}' + --filter '{"field":"status","operator":"eq","value":"published"}' # Composed conditions — works with search, vsearch, and query qmd query "dependency injection" --filter '{ "operator": "and", "operands": [ - { "key": "topics", "operator": "all", "value": ["typescript", "programming"] }, - { "key": "status", "operator": "nin", "value": ["draft", "archived"] }, + { "field": "topics", "operator": "all", "value": ["typescript", "programming"] }, + { "field": "status", "operator": "nin", "value": ["draft", "archived"] }, { "operator": "or", "operands": [ - { "key": "priority", "operator": "gte", "value": 3 }, - { "key": "reviewed", "operator": "eq", "value": true } + { "field": "priority", "operator": "gte", "value": 3 }, + { "field": "reviewed", "operator": "eq", "value": true } ] }, - { "operator": "not", "operand": { "key": "audience", "operator": "eq", "value": "internal" } } + { "operator": "not", "operand": { "field": "audience", "operator": "eq", "value": "internal" } } ] }' ``` @@ -966,9 +966,9 @@ qmd query "dependency injection" --filter '{ |------|-------| | Logical group | `{ "operator": "and" \| "or", "operands": […] }` | | Negation | `{ "operator": "not", "operand": {…} }` | -| Comparison | `{ "key", "operator": "eq" \| "ne" \| "gt" \| "gte" \| "lt" \| "lte", "value" }` | -| Membership | `{ "key", "operator": "in" \| "nin" \| "all", "value": […] }` | -| Presence | `{ "key", "operator": "exists", "value": true \| false }` | +| Comparison | `{ "field", "operator": "eq" \| "ne" \| "gt" \| "gte" \| "lt" \| "lte", "value" }` | +| Membership | `{ "field", "operator": "in" \| "nin" \| "all", "value": […] }` | +| Presence | `{ "field", "operator": "exists", "value": true \| false }` | Semantics: diff --git a/skills/qmd/SKILL.md b/skills/qmd/SKILL.md index 7693d4273..68dbbbb80 100644 --- a/skills/qmd/SKILL.md +++ b/skills/qmd/SKILL.md @@ -204,11 +204,11 @@ Omit `-c` to search everything. Documents can carry typed metadata in a `qmd.metadata` frontmatter block (strings, numbers, booleans, or flat arrays). `search`, `vsearch`, and `query` accept `--filter` with a recursive JSON AST; every returned result satisfies it: ```bash -qmd search "authentication" --filter '{"key":"status","operator":"eq","value":"published"}' -qmd query "dependency injection" --filter '{"operator":"and","operands":[{"key":"topics","operator":"all","value":["typescript"]},{"key":"status","operator":"nin","value":["draft","archived"]}]}' +qmd search "authentication" --filter '{"field":"status","operator":"eq","value":"published"}' +qmd query "dependency injection" --filter '{"operator":"and","operands":[{"field":"topics","operator":"all","value":["typescript"]},{"field":"status","operator":"nin","value":["draft","archived"]}]}' ``` -Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` takes one `operand`, and conditions take `key` + `value` with operators `eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), or `exists` (presence). Matching is typed and exact; missing keys do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to include them). The MCP `query` tool accepts the same AST as a `filter` object. JSON output includes each result's `metadata`. +Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` takes one `operand`, and conditions take `field` + `value` with operators `eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), or `exists` (presence). Matching is typed and exact; missing keys do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to include them). The MCP `query` tool accepts the same AST as a `filter` object. JSON output includes each result's `metadata`. ## MCP Tool: `query` diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index c5caa2668..7e386b954 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -2832,7 +2832,7 @@ function parseCliMetadataFilter(rawFilter: unknown): MetadataFilter | undefined filterJson = JSON.parse(String(rawFilter)); } catch (err) { console.error(`Invalid --filter JSON: ${err instanceof Error ? err.message : String(err)}`); - console.error(`Example: --filter '{"key":"status","operator":"eq","value":"published"}'`); + console.error(`Example: --filter '{"field":"status","operator":"eq","value":"published"}'`); process.exit(1); } @@ -3730,7 +3730,7 @@ function showHelp(): void { console.log(" --format - Output format: cli (default) | json | csv | md | xml | files"); console.log(" -c, --collection - Filter by one or more collections"); console.log(" --filter - Metadata filter (recursive JSON AST; search/vsearch/query)"); - console.log(" e.g. '{\"key\":\"status\",\"operator\":\"eq\",\"value\":\"published\"}'"); + console.log(" e.g. '{\"field\":\"status\",\"operator\":\"eq\",\"value\":\"published\"}'"); console.log(""); console.log("Embed/query options:"); console.log(" --chunk-strategy - Chunking mode (default: regex; auto uses AST for code files)"); diff --git a/src/mcp/server.ts b/src/mcp/server.ts index 73f3ee460..c92efc110 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -348,11 +348,11 @@ Intent-aware lex (C++ performance, not sports): filter: z.record(z.string(), z.unknown()).optional().describe( "Metadata filter (recursive JSON AST). Every returned result satisfies it. " + "Nodes are operator-discriminated: logical groups {operator:'and'|'or', operands:[...]}, " + - "negation {operator:'not', operand:{...}}, and conditions {key, operator, value} with " + + "negation {operator:'not', operand:{...}}, and conditions {field, operator, value} with " + "operators eq/ne/gt/gte/lt/lte (comparison), in/nin/all (membership), exists (presence). " + "Values are typed exactly (no coercion); missing keys do not match ne/nin. " + - "Example: {\"operator\":\"and\",\"operands\":[{\"key\":\"topics\",\"operator\":\"all\",\"value\":[\"typescript\"]}," + - "{\"key\":\"status\",\"operator\":\"ne\",\"value\":\"draft\"}]}" + "Example: {\"operator\":\"and\",\"operands\":[{\"field\":\"topics\",\"operator\":\"all\",\"value\":[\"typescript\"]}," + + "{\"field\":\"status\",\"operator\":\"ne\",\"value\":\"draft\"}]}" ), intent: z.string().optional().describe( "Background context to disambiguate the query. Example: query='performance', intent='web page load times and Core Web Vitals'. Does not search on its own." diff --git a/src/metadata-filter.ts b/src/metadata-filter.ts index aa16e7cd6..5456aeff8 100644 --- a/src/metadata-filter.ts +++ b/src/metadata-filter.ts @@ -7,7 +7,11 @@ * * { "operator": "and", "operands": [ ... ] } * { "operator": "not", "operand": { ... } } - * { "key": "status", "operator": "eq", "value": "published" } + * { "field": "status", "operator": "eq", "value": "published" } + * + * A condition is a predicate over one field of the record under evaluation: + * `field` names it, `operator` says how to compare, `value` is the operand. + * For a document, the fields are its metadata keys. * * Compilation emits correlated EXISTS/NOT EXISTS subqueries over * `document_metadata_values` with every user value bound as a parameter — @@ -37,10 +41,10 @@ export interface MetadataFilterNegation { } export type MetadataCondition = - | { key: string; operator: "eq" | "ne"; value: MetadataScalar } - | { key: string; operator: "gt" | "gte" | "lt" | "lte"; value: string | number } - | { key: string; operator: "in" | "nin" | "all"; value: MetadataScalarArray } - | { key: string; operator: "exists"; value: boolean }; + | { field: string; operator: "eq" | "ne"; value: MetadataScalar } + | { field: string; operator: "gt" | "gte" | "lt" | "lte"; value: string | number } + | { field: string; operator: "in" | "nin" | "all"; value: MetadataScalarArray } + | { field: string; operator: "exists"; value: boolean }; export interface CompiledMetadataFilter { sql: string; @@ -175,14 +179,14 @@ function parseFilterNegation( } function parseFilterCondition(node: Record, operator: string, path: string): MetadataCondition { - rejectUnknownProperties(node, ["key", "operator", "value"], path); + rejectUnknownProperties(node, ["field", "operator", "value"], path); - const key = node["key"]; - if (typeof key !== "string" || key.length === 0) { - throw new MetadataFilterError(path, `'${operator}' requires a non-empty string 'key'`); + const field = node["field"]; + if (typeof field !== "string" || field.length === 0) { + throw new MetadataFilterError(path, `'${operator}' requires a non-empty string 'field'`); } - if (Buffer.byteLength(key, "utf-8") > METADATA_FILTER_LIMITS.maxKeyBytes) { - throw new MetadataFilterError(path, `'key' exceeds ${METADATA_FILTER_LIMITS.maxKeyBytes} bytes`); + if (Buffer.byteLength(field, "utf-8") > METADATA_FILTER_LIMITS.maxKeyBytes) { + throw new MetadataFilterError(path, `'field' exceeds ${METADATA_FILTER_LIMITS.maxKeyBytes} bytes`); } if (!("value" in node)) { @@ -194,12 +198,12 @@ function parseFilterCondition(node: Record, operator: string, p if (typeof value !== "boolean") { throw new MetadataFilterError(`${path}.value`, "'exists' requires a boolean value"); } - return { key, operator, value }; + return { field, operator, value }; } if (MEMBERSHIP_OPERATORS.has(operator)) { return { - key, + field, operator: operator as "in" | "nin" | "all", value: parseMembershipValues(value, operator, path), }; @@ -210,7 +214,7 @@ function parseFilterCondition(node: Record, operator: string, p if (ORDERED_OPERATORS.has(operator) && typeof scalar === "boolean") { throw new MetadataFilterError(`${path}.value`, `'${operator}' requires a string or number value`); } - return { key, operator, value: scalar } as MetadataCondition; + return { field, operator, value: scalar } as MetadataCondition; } function parseMembershipValues(value: unknown, operator: string, path: string): MetadataScalarArray { @@ -293,7 +297,7 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin return `NOT ${compileFilterNode(filter.operand, alias, params)}`; case "exists": - params.push(filter.key); + params.push(filter.field); return filter.value ? buildValueExistsSql(alias, "mv.key = ?") : `NOT ${buildValueExistsSql(alias, "mv.key = ?")}`; @@ -304,7 +308,7 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin case "lt": case "lte": { const sqlOperator = { eq: "=", gt: ">", gte: ">=", lt: "<", lte: "<=" }[filter.operator]; - params.push(filter.key, bindScalar(filter.value)); + params.push(filter.field, bindScalar(filter.value)); return buildValueExistsSql( alias, `mv.key = ? AND mv.value_type = '${valueTypeOf(filter.value)}' AND mv.${valueColumnOf(filter.value)} ${sqlOperator} ?`, @@ -315,9 +319,9 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin // Key must have at least one same-type value, and no same-type value // may equal the operand. Missing keys and type mismatches do not match. const valueType = valueTypeOf(filter.value); - params.push(filter.key); + params.push(filter.field); const presentSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}'`); - params.push(filter.key, bindScalar(filter.value)); + params.push(filter.field, bindScalar(filter.value)); const equalSql = buildValueExistsSql( alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${valueColumnOf(filter.value)} = ?`, @@ -332,13 +336,13 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin const placeholders = filter.value.map(() => "?").join(", "); if (filter.operator === "in") { - params.push(filter.key, ...filter.value.map(bindScalar)); + params.push(filter.field, ...filter.value.map(bindScalar)); return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} IN (${placeholders})`); } - params.push(filter.key); + params.push(filter.field); const presentSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}'`); - params.push(filter.key, ...filter.value.map(bindScalar)); + params.push(filter.field, ...filter.value.map(bindScalar)); const memberSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} IN (${placeholders})`); return `(${presentSql} AND NOT ${memberSql})`; } @@ -347,7 +351,7 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin const valueType = valueTypeOf(filter.value[0]!); const column = valueColumnOf(filter.value[0]!); const memberSqls = filter.value.map(element => { - params.push(filter.key, bindScalar(element)); + params.push(filter.field, bindScalar(element)); return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} = ?`); }); return `(${memberSqls.join(" AND ")})`; diff --git a/test/metadata-cli.test.ts b/test/metadata-cli.test.ts index 213f0eede..5334fed34 100644 --- a/test/metadata-cli.test.ts +++ b/test/metadata-cli.test.ts @@ -93,7 +93,7 @@ describe("qmd search --filter", () => { const { stdout, exitCode } = await runQmd([ "search", "cli filter keyword", "--format", "json", - "--filter", '{"key":"status","operator":"eq","value":"published"}', + "--filter", '{"field":"status","operator":"eq","value":"published"}', ]); expect(exitCode).toBe(0); @@ -107,8 +107,8 @@ describe("qmd search --filter", () => { const filter = JSON.stringify({ operator: "and", operands: [ - { key: "topics", operator: "all", value: ["typescript", "programming"] }, - { operator: "not", operand: { key: "status", operator: "eq", value: "draft" } }, + { field: "topics", operator: "all", value: ["typescript", "programming"] }, + { operator: "not", operand: { field: "status", operator: "eq", value: "draft" } }, ], }); const { stdout, exitCode } = await runQmd([ @@ -122,7 +122,7 @@ describe("qmd search --filter", () => { const { stdout, exitCode } = await runQmd([ "search", "cli filter keyword", "--format", "json", - "--filter", '{"key":"status","operator":"eq","value":"missing"}', + "--filter", '{"field":"status","operator":"eq","value":"missing"}', ]); expect(exitCode).toBe(0); expect(JSON.parse(stdout)).toEqual([]); @@ -152,7 +152,7 @@ describe("qmd search --filter", () => { test("rejects valid JSON with an invalid filter AST", async () => { const { stderr, exitCode } = await runQmd([ - "search", "cli filter keyword", "--filter", '{"key":"status","operator":"equal","value":"x"}', + "search", "cli filter keyword", "--filter", '{"field":"status","operator":"equal","value":"x"}', ]); expect(exitCode).toBe(1); expect(stderr).toMatch(/Invalid metadata filter at \$/); diff --git a/test/metadata-filter.test.ts b/test/metadata-filter.test.ts index 6e36e921e..0d1c883fd 100644 --- a/test/metadata-filter.test.ts +++ b/test/metadata-filter.test.ts @@ -23,16 +23,16 @@ import { METADATA_EXTRACTION_VERSION, type DocumentMetadata } from "../src/metad describe("parseMetadataFilter", () => { test("accepts every condition operator shape", () => { const conditions: unknown[] = [ - { key: "status", operator: "eq", value: "published" }, - { key: "status", operator: "ne", value: "draft" }, - { key: "priority", operator: "gt", value: 3 }, - { key: "priority", operator: "gte", value: 3 }, - { key: "priority", operator: "lt", value: 10 }, - { key: "name", operator: "lte", value: "m" }, - { key: "topics", operator: "in", value: ["a", "b"] }, - { key: "topics", operator: "nin", value: [1, 2] }, - { key: "flags", operator: "all", value: [true, false] }, - { key: "status", operator: "exists", value: false }, + { field: "status", operator: "eq", value: "published" }, + { field: "status", operator: "ne", value: "draft" }, + { field: "priority", operator: "gt", value: 3 }, + { field: "priority", operator: "gte", value: 3 }, + { field: "priority", operator: "lt", value: 10 }, + { field: "name", operator: "lte", value: "m" }, + { field: "topics", operator: "in", value: ["a", "b"] }, + { field: "topics", operator: "nin", value: [1, 2] }, + { field: "flags", operator: "all", value: [true, false] }, + { field: "status", operator: "exists", value: false }, ]; for (const condition of conditions) { expect(parseMetadataFilter(condition)).toEqual(condition); @@ -43,12 +43,12 @@ describe("parseMetadataFilter", () => { const filter = { operator: "and", operands: [ - { key: "topics", operator: "all", value: ["typescript", "programming"] }, + { field: "topics", operator: "all", value: ["typescript", "programming"] }, { operator: "or", operands: [ - { key: "status", operator: "eq", value: "published" }, - { operator: "not", operand: { key: "audience", operator: "eq", value: "internal" } }, + { field: "status", operator: "eq", value: "published" }, + { operator: "not", operand: { field: "audience", operator: "eq", value: "internal" } }, ], }, ], @@ -57,31 +57,32 @@ describe("parseMetadataFilter", () => { }); test("canonicalizes membership arrays by de-duplicating", () => { - const parsed = parseMetadataFilter({ key: "topics", operator: "in", value: ["a", "b", "a"] }); - expect(parsed).toEqual({ key: "topics", operator: "in", value: ["a", "b"] }); + const parsed = parseMetadataFilter({ field: "topics", operator: "in", value: ["a", "b", "a"] }); + expect(parsed).toEqual({ field: "topics", operator: "in", value: ["a", "b"] }); }); test("rejects invalid node shapes with the failing JSON path", () => { const cases: [unknown, RegExp][] = [ ["not-an-object", /at \$:.*must be an object/], - [{ key: "a", value: 1 }, /at \$:.*missing 'operator'/], - [{ operator: "equal", key: "a", value: 1 }, /unknown operator 'equal'/], + [{ field: "a", value: 1 }, /at \$:.*missing 'operator'/], + [{ operator: "equal", field: "a", value: 1 }, /unknown operator 'equal'/], [{ operator: "and", operands: [] }, /non-empty 'operands'/], [{ operator: "and", operands: "nope" }, /'operands' array/], - [{ operator: "not", operands: [{ key: "a", operator: "eq", value: 1 }] }, /unknown property 'operands'/], + [{ operator: "not", operands: [{ field: "a", operator: "eq", value: 1 }] }, /unknown property 'operands'/], [{ operator: "not" }, /exactly one 'operand'/], [{ operator: "and", operands: [{ operator: "eq" }] }, /at \$\.operands\[0\]/], - [{ operator: "eq", value: 1 }, /non-empty string 'key'/], - [{ operator: "eq", key: "a" }, /requires a 'value'/], - [{ operator: "eq", key: "a", value: 1, extra: true }, /unknown property 'extra'/], - [{ operator: "and", operands: [{ key: "a", operator: "eq", value: 1 }], key: "a" }, /unknown property 'key'/], - [{ operator: "gt", key: "a", value: true }, /string or number value/], - [{ operator: "eq", key: "a", value: NaN }, /finite/], - [{ operator: "eq", key: "a", value: { nested: 1 } }, /string, number, or boolean/], - [{ operator: "in", key: "a", value: "x" }, /array value/], - [{ operator: "in", key: "a", value: [] }, /non-empty array/], - [{ operator: "in", key: "a", value: [1, "two"] }, /homogeneous array/], - [{ operator: "exists", key: "a", value: "yes" }, /boolean value/], + [{ operator: "eq", value: 1 }, /non-empty string 'field'/], + [{ operator: "eq", field: "a" }, /requires a 'value'/], + [{ operator: "eq", field: "a", value: 1, extra: true }, /unknown property 'extra'/], + [{ operator: "and", operands: [{ field: "a", operator: "eq", value: 1 }], field: "a" }, /unknown property 'field'/], + [{ operator: "eq", key: "a", value: 1 }, /unknown property 'key'/], + [{ operator: "gt", field: "a", value: true }, /string or number value/], + [{ operator: "eq", field: "a", value: NaN }, /finite/], + [{ operator: "eq", field: "a", value: { nested: 1 } }, /string, number, or boolean/], + [{ operator: "in", field: "a", value: "x" }, /array value/], + [{ operator: "in", field: "a", value: [] }, /non-empty array/], + [{ operator: "in", field: "a", value: [1, "two"] }, /homogeneous array/], + [{ operator: "exists", field: "a", value: "yes" }, /boolean value/], ]; for (const [input, expected] of cases) { @@ -91,13 +92,13 @@ describe("parseMetadataFilter", () => { }); test("rejects excessive depth, node count, and operand count", () => { - let deepFilter: unknown = { key: "a", operator: "eq", value: 1 }; + let deepFilter: unknown = { field: "a", operator: "eq", value: 1 }; for (let i = 0; i <= METADATA_FILTER_LIMITS.maxDepth; i++) { deepFilter = { operator: "not", operand: deepFilter }; } expect(() => parseMetadataFilter(deepFilter)).toThrow(/nesting depth/); - const condition = { key: "a", operator: "eq", value: 1 }; + const condition = { field: "a", operator: "eq", value: 1 }; const wideGroup = { operator: "or", operands: Array.from({ length: METADATA_FILTER_LIMITS.maxGroupOperands + 1 }, () => condition), @@ -114,7 +115,7 @@ describe("parseMetadataFilter", () => { expect(() => parseMetadataFilter(manyNodes)).toThrow(/nodes/); const manyValues = { - key: "a", + field: "a", operator: "in", value: Array.from({ length: METADATA_FILTER_LIMITS.maxMembershipValues + 1 }, (_, i) => i), }; @@ -170,10 +171,10 @@ describe("compileMetadataFilter semantics", () => { insertDoc("num.md", { status: 1 }); insertDoc("bool.md", { status: true }); - expect(matchPaths({ key: "status", operator: "eq", value: "published" })).toEqual(["str.md"]); - expect(matchPaths({ key: "status", operator: "eq", value: 1 })).toEqual(["num.md"]); - expect(matchPaths({ key: "status", operator: "eq", value: true })).toEqual(["bool.md"]); - expect(matchPaths({ key: "status", operator: "eq", value: "1" })).toEqual([]); + expect(matchPaths({ field: "status", operator: "eq", value: "published" })).toEqual(["str.md"]); + expect(matchPaths({ field: "status", operator: "eq", value: 1 })).toEqual(["num.md"]); + expect(matchPaths({ field: "status", operator: "eq", value: true })).toEqual(["bool.md"]); + expect(matchPaths({ field: "status", operator: "eq", value: "1" })).toEqual([]); }); test("ne requires key presence and matching type", () => { @@ -182,7 +183,7 @@ describe("compileMetadataFilter semantics", () => { insertDoc("missing.md", { other: "x" }); insertDoc("typed.md", { status: 1 }); - expect(matchPaths({ key: "status", operator: "ne", value: "draft" })).toEqual(["published.md"]); + expect(matchPaths({ field: "status", operator: "ne", value: "draft" })).toEqual(["published.md"]); }); test("ordered comparisons match numbers and binary-ordered strings", () => { @@ -192,11 +193,11 @@ describe("compileMetadataFilter semantics", () => { insertDoc("alpha.md", { name: "alpha" }); insertDoc("zulu.md", { name: "zulu" }); - expect(matchPaths({ key: "priority", operator: "gte", value: 3 })).toEqual(["high.md", "mid.md"]); - expect(matchPaths({ key: "priority", operator: "lt", value: 3 })).toEqual(["low.md"]); - expect(matchPaths({ key: "name", operator: "gt", value: "alpha" })).toEqual(["zulu.md"]); + expect(matchPaths({ field: "priority", operator: "gte", value: 3 })).toEqual(["high.md", "mid.md"]); + expect(matchPaths({ field: "priority", operator: "lt", value: 3 })).toEqual(["low.md"]); + expect(matchPaths({ field: "name", operator: "gt", value: "alpha" })).toEqual(["zulu.md"]); // Type mismatch: no string 'priority' values exist. - expect(matchPaths({ key: "priority", operator: "gte", value: "3" })).toEqual([]); + expect(matchPaths({ field: "priority", operator: "gte", value: "3" })).toEqual([]); }); test("in, nin, and all evaluate membership over value sets", () => { @@ -204,18 +205,18 @@ describe("compileMetadataFilter semantics", () => { insertDoc("go.md", { topics: ["go"] }); insertDoc("none.md", { other: "x" }); - expect(matchPaths({ key: "topics", operator: "in", value: ["typescript", "rust"] })).toEqual(["ts.md"]); - expect(matchPaths({ key: "topics", operator: "nin", value: ["typescript", "rust"] })).toEqual(["go.md"]); - expect(matchPaths({ key: "topics", operator: "all", value: ["typescript", "programming"] })).toEqual(["ts.md"]); - expect(matchPaths({ key: "topics", operator: "all", value: ["typescript", "rust"] })).toEqual([]); + expect(matchPaths({ field: "topics", operator: "in", value: ["typescript", "rust"] })).toEqual(["ts.md"]); + expect(matchPaths({ field: "topics", operator: "nin", value: ["typescript", "rust"] })).toEqual(["go.md"]); + expect(matchPaths({ field: "topics", operator: "all", value: ["typescript", "programming"] })).toEqual(["ts.md"]); + expect(matchPaths({ field: "topics", operator: "all", value: ["typescript", "rust"] })).toEqual([]); }); test("exists matches presence and absence", () => { insertDoc("has.md", { status: "ok" }); insertDoc("hasnt.md", { other: "x" }); - expect(matchPaths({ key: "status", operator: "exists", value: true })).toEqual(["has.md"]); - expect(matchPaths({ key: "status", operator: "exists", value: false })).toEqual(["hasnt.md"]); + expect(matchPaths({ field: "status", operator: "exists", value: true })).toEqual(["has.md"]); + expect(matchPaths({ field: "status", operator: "exists", value: false })).toEqual(["hasnt.md"]); }); test("and, or, and not compose recursively", () => { @@ -226,22 +227,22 @@ describe("compileMetadataFilter semantics", () => { expect(matchPaths({ operator: "and", operands: [ - { key: "status", operator: "eq", value: "published" }, - { key: "priority", operator: "gte", value: 3 }, + { field: "status", operator: "eq", value: "published" }, + { field: "priority", operator: "gte", value: 3 }, ], })).toEqual(["a.md"]); expect(matchPaths({ operator: "or", operands: [ - { key: "priority", operator: "gte", value: 9 }, - { key: "priority", operator: "lte", value: 1 }, + { field: "priority", operator: "gte", value: 9 }, + { field: "priority", operator: "lte", value: 1 }, ], })).toEqual(["b.md", "c.md"]); expect(matchPaths({ operator: "not", - operand: { key: "status", operator: "eq", value: "draft" }, + operand: { field: "status", operator: "eq", value: "draft" }, })).toEqual(["a.md", "b.md"]); }); @@ -252,8 +253,8 @@ describe("compileMetadataFilter semantics", () => { const rangeFilter: MetadataFilter = { operator: "and", operands: [ - { key: "priority", operator: "gte", value: 3 }, - { key: "priority", operator: "lt", value: 10 }, + { field: "priority", operator: "gte", value: 3 }, + { field: "priority", operator: "lt", value: 10 }, ], }; // Document-level semantics: [1, 20] satisfies both conditions via @@ -265,15 +266,15 @@ describe("compileMetadataFilter semantics", () => { insertDoc("flagged.md", { reviewed: true }); insertDoc("unflagged.md", { reviewed: false }); - expect(matchPaths({ key: "reviewed", operator: "in", value: [true] })).toEqual(["flagged.md"]); - expect(matchPaths({ key: "reviewed", operator: "nin", value: [true] })).toEqual(["unflagged.md"]); + expect(matchPaths({ field: "reviewed", operator: "in", value: [true] })).toEqual(["flagged.md"]); + expect(matchPaths({ field: "reviewed", operator: "nin", value: [true] })).toEqual(["unflagged.md"]); }); test("SQL injection payloads in keys and values stay data", () => { insertDoc("safe.md", { "key'; DROP TABLE documents; --": "v'; DROP TABLE documents; --" }); expect(matchPaths({ - key: "key'; DROP TABLE documents; --", + field: "key'; DROP TABLE documents; --", operator: "eq", value: "v'; DROP TABLE documents; --", })).toEqual(["safe.md"]); @@ -286,8 +287,8 @@ describe("compileMetadataFilter semantics", () => { const filter = parseMetadataFilter({ operator: "and", operands: [ - { key: "key'; --", operator: "eq", value: "value'; --" }, - { key: "topics", operator: "in", value: ["a'; --"] }, + { field: "key'; --", operator: "eq", value: "value'; --" }, + { field: "topics", operator: "in", value: ["a'; --"] }, ], }); const compiled = compileMetadataFilter(filter, "d"); diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index 6e4b1741b..fbcc0fb7a 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -71,7 +71,7 @@ describe("searchFTS with metadata filter", () => { expect(unfiltered.length).toBe(3); const filtered = searchFTS(store.db, "authentication", 10, undefined, { - key: "status", operator: "eq", value: "published", + field: "status", operator: "eq", value: "published", }); expect(filtered.map(r => r.displayPath)).toEqual(["notes/published.md"]); expect(filtered[0]!.metadata).toEqual({ status: "published" }); @@ -95,7 +95,7 @@ describe("searchFTS with metadata filter", () => { // An unprocessed document must not satisfy `exists: false`. const filtered = searchFTS(store.db, "common", 10, undefined, { - key: "status", operator: "exists", value: false, + field: "status", operator: "exists", value: false, }); expect(filtered.map(r => r.displayPath)).toEqual(["notes/extracted.md"]); @@ -110,7 +110,7 @@ describe("searchFTS with metadata filter", () => { await insertDoc("other", "d.md", "# D\n\nshared topic", { status: "published" }); const filtered = searchFTS(store.db, "shared", 10, ["notes", "docs"], { - key: "status", operator: "eq", value: "published", + field: "status", operator: "eq", value: "published", }); expect(filtered.map(r => r.displayPath).sort()).toEqual(["docs/b.md", "notes/a.md"]); }); @@ -129,12 +129,12 @@ describe("searchFTS with metadata filter", () => { const filter: MetadataFilter = { operator: "and", operands: [ - { key: "topics", operator: "all", value: ["typescript", "programming"] }, + { field: "topics", operator: "all", value: ["typescript", "programming"] }, { operator: "or", operands: [ - { key: "status", operator: "eq", value: "published" }, - { key: "priority", operator: "gte", value: 3 }, + { field: "status", operator: "eq", value: "published" }, + { field: "priority", operator: "gte", value: 3 }, ], }, ], @@ -151,7 +151,7 @@ describe("searchFTS with metadata filter", () => { } const filtered = searchFTS(store.db, "repeated keyword", 5, undefined, { - key: "status", operator: "eq", value: "published", + field: "status", operator: "eq", value: "published", }); expect(filtered.map(r => r.displayPath)).toEqual(["notes/doc-0.md"]); }); @@ -182,7 +182,7 @@ describe("searchVec with metadata filter", () => { const filtered = await searchVec( store.db, "q", model, 10, undefined, undefined, queryEmbedding, undefined, - { key: "status", operator: "eq", value: "published" }, + { field: "status", operator: "eq", value: "published" }, ); expect(filtered.map(r => r.displayPath)).toEqual(["notes/far-published.md"]); expect(filtered[0]!.metadata).toEqual({ status: "published" }); @@ -194,7 +194,7 @@ describe("searchVec with metadata filter", () => { const filtered = await searchVec( store.db, "q", model, 10, undefined, undefined, queryEmbedding, undefined, - { key: "status", operator: "eq", value: "published" }, + { field: "status", operator: "eq", value: "published" }, ); expect(filtered).toEqual([]); }); @@ -218,7 +218,7 @@ describe("searchVec with metadata filter", () => { const filtered = await searchVec( store.db, "q", model, 10, undefined, undefined, queryEmbedding, undefined, - { key: "status", operator: "eq", value: "published" }, + { field: "status", operator: "eq", value: "published" }, ); expect(filtered.map(r => r.displayPath)).toEqual(["notes/published-copy.md"]); }); @@ -230,7 +230,7 @@ describe("searchVec with metadata filter", () => { const filtered = await searchVec( store.db, "q", model, 10, "docs", undefined, queryEmbedding, undefined, - { key: "status", operator: "eq", value: "published" }, + { field: "status", operator: "eq", value: "published" }, ); expect(filtered.map(r => r.displayPath)).toEqual(["docs/b.md"]); }); @@ -248,7 +248,7 @@ describe("structuredSearch with metadata filter", () => { syncDocumentMetadata(store.db, draftId, draftBody, "draft.md"); const results = await structuredSearch(store, [{ type: "lex", query: "structured keyword" }], { - filter: { key: "status", operator: "eq", value: "published" }, + filter: { field: "status", operator: "eq", value: "published" }, skipRerank: true, }); diff --git a/test/metadata-surfaces.test.ts b/test/metadata-surfaces.test.ts index 3b5eec3cc..14ff10eac 100644 --- a/test/metadata-surfaces.test.ts +++ b/test/metadata-surfaces.test.ts @@ -68,7 +68,7 @@ describe("SDK metadata filter", () => { expect(unfiltered.length).toBe(2); const filtered = await store.searchLex("sdk keyword", { - filter: { key: "status", operator: "eq", value: "published" }, + filter: { field: "status", operator: "eq", value: "published" }, }); expect(filtered.map(r => r.displayPath)).toEqual(["docs/published.md"]); expect(filtered[0]!.metadata).toEqual({ status: "published", topics: ["typescript"] }); @@ -77,7 +77,7 @@ describe("SDK metadata filter", () => { test("search with pre-expanded queries applies the filter", async () => { const results = await store.search({ queries: [{ type: "lex", query: "sdk keyword" }], - filter: { key: "status", operator: "ne", value: "draft" }, + filter: { field: "status", operator: "ne", value: "draft" }, rerank: false, }); expect(results.map(r => r.displayPath)).toEqual(["docs/published.md"]); @@ -190,7 +190,7 @@ describe("MCP and HTTP metadata filter", () => { test("POST /query applies the filter and includes metadata", async () => { const { status, json } = await postJson("/query", { searches: [{ type: "lex", query: "http keyword" }], - filter: { key: "status", operator: "eq", value: "published" }, + filter: { field: "status", operator: "eq", value: "published" }, rerank: false, }); expect(status).toBe(200); @@ -202,7 +202,7 @@ describe("MCP and HTTP metadata filter", () => { test("POST /search alias accepts the same filter", async () => { const { status, json } = await postJson("/search", { searches: [{ type: "lex", query: "http keyword" }], - filter: { key: "status", operator: "eq", value: "draft" }, + filter: { field: "status", operator: "eq", value: "draft" }, rerank: false, }); expect(status).toBe(200); @@ -220,7 +220,7 @@ describe("MCP and HTTP metadata filter", () => { const invalidAst = await postJson("/query", { searches: [{ type: "lex", query: "http keyword" }], - filter: { key: "status", operator: "equal", value: "published" }, + filter: { field: "status", operator: "equal", value: "published" }, }); expect(invalidAst.status).toBe(400); expect(invalidAst.json.error).toMatch(/unknown operator 'equal'/); @@ -232,8 +232,8 @@ describe("MCP and HTTP metadata filter", () => { filter: { operator: "and", operands: [ - { key: "status", operator: "eq", value: "published" }, - { operator: "not", operand: { key: "status", operator: "eq", value: "draft" } }, + { field: "status", operator: "eq", value: "published" }, + { operator: "not", operand: { field: "status", operator: "eq", value: "draft" } }, ], }, rerank: false, From 21a0b289779e62558bf5c2bc70c693852cc004de Mon Sep 17 00:00:00 2001 From: Rik van Riel Date: Wed, 16 Sep 2026 14:20:59 -0400 Subject: [PATCH 14/82] feat(index): add mtime+size fast-path via file_sync_state When a collection has no changes, reindexing still reads every file to hash it. With thousands of files this costs minutes of IO and hashing even when nothing changed. Add file_sync_state table tracking mtime_ms, size, and content hash per (collection, path). On reindex, stat the file first and skip the read entirely when mtime and size match the cache. If mtime changed but hash is identical only the cached mtime is updated. Skip files over 10MB and empty files, and remove the cache entry on orphan deletion. Second reindex with no changes performs no file reads, taking update from minutes to seconds when the collection is unchanged. (cherry picked from commit 3fcf2307a665f4da764095a28582016c747099db) --- src/store.ts | 174 +++++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 154 insertions(+), 20 deletions(-) diff --git a/src/store.ts b/src/store.ts index cf532d7c8..bb76647b4 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1285,6 +1285,21 @@ function initializeDatabase(db: Database): void { ) `); + // File sync state — mtime+size fast-path. + // Inspired by qmd-py incremental sync. + db.exec(` + CREATE TABLE IF NOT EXISTS file_sync_state ( + collection TEXT NOT NULL, + relative_path TEXT NOT NULL, + mtime_ms INTEGER NOT NULL, + size INTEGER NOT NULL, + content_hash TEXT NOT NULL, + document_id INTEGER NOT NULL, + PRIMARY KEY (collection, relative_path) + ) + `); + db.exec(`CREATE INDEX IF NOT EXISTS idx_file_sync_state_collection ON file_sync_state(collection)`); + // FTS - index filepath (collection/path), title, and content. // Do not CREATE VIRTUAL TABLE here as an autocommit statement: FTS5 // IF NOT EXISTS races under WAL (see createDocumentsFtsTable). @@ -1613,9 +1628,68 @@ export type ReindexResult = { metadataErrors: number; }; +/** + * File sync state row — mtime+size fast-path cache. + * Inspired by qmd-py incremental sync. + */ +type FileSyncStateRow = { + relative_path: string; + mtime_ms: number; + size: number; + content_hash: string; + document_id: number; +}; + +function getFileSyncStateMap(db: Database, collectionName: string): Map { + try { + const rows = db.prepare( + `SELECT relative_path, mtime_ms, size, content_hash, document_id FROM file_sync_state WHERE collection = ?` + ).all(collectionName) as FileSyncStateRow[]; + const map = new Map(); + for (const r of rows) map.set(r.relative_path, r); + return map; + } catch { + // Table may not exist yet on legacy DBs — initializeDatabase creates it on next open, + // but guard here for safety. + return new Map(); + } +} + +function upsertFileSyncState(db: Database, collectionName: string, relPath: string, mtimeMs: number, size: number, contentHash: string, documentId: number): void { + try { + db.prepare(` + INSERT INTO file_sync_state(collection, relative_path, mtime_ms, size, content_hash, document_id) + VALUES (?, ?, ?, ?, ?, ?) + ON CONFLICT(collection, relative_path) DO UPDATE SET + mtime_ms = excluded.mtime_ms, + size = excluded.size, + content_hash = excluded.content_hash, + document_id = excluded.document_id + `).run(collectionName, relPath, Math.floor(mtimeMs), size, contentHash, documentId); + } catch { + // Legacy DB without table — will be created on next open; skip caching this run + } +} + +function deleteFileSyncStateForCollection(db: Database, collectionName: string, relPath: string): void { + try { + db.prepare(`DELETE FROM file_sync_state WHERE collection = ? AND relative_path = ?`).run(collectionName, relPath); + } catch {} +} + +/** + * Maximum file size to index — prevents OOM on accidental binary inclusion. + */ +const REINDEX_MAX_FILE_SIZE = 10 * 1024 * 1024; // 10MB + /** * Re-index a single collection by scanning the filesystem and updating the database. + * Uses mtime+size fast-path (file_sync_state) to avoid re-reading unchanged files. * Pure function — no console output, no db lifecycle management. + * + * Fast-path: stat mtime_ms+size against cached row to skip file read. + * If mtime changed but content hash identical, only mtime cache is updated. + * Skips >10MB and empty files, cleans sync table entry on orphan removal. */ export async function reindexCollection( store: Store, @@ -1652,19 +1726,14 @@ export async function reindexCollection( let indexed = 0, updated = 0, unchanged = 0, processed = 0, metadataErrors = 0; const skippedFiles: ReindexSkippedFile[] = []; const seenPaths = new Set(); - // Literal paths of every file in this scan. Passed to the legacy-path - // migration so it never adopts a row that still belongs to a live file. const livePaths = new Set(files.map(f => normalizePathSeparators(f))); + // Load file_sync_state for this collection (mtime+size fast-path) + const syncStateMap = getFileSyncStateMap(db, collectionName); + for (const relativeFile of files) { - const filepath = getRealPath(resolve(collectionPath, relativeFile)); - // Store the literal relative path so the filesystem path can always be - // reconstructed as: resolve(collection.path, storedPath). - // handelize() is NOT applied at index time — it is display-only. const path = normalizePathSeparators(relativeFile); - // Glob `../` segments, absolute patterns, and file symlinks can resolve - // outside the collection root. Do not ingest those files, and do not mark - // them seen so a previous escaped row is deactivated on this pass. + const filepath = getRealPath(resolve(collectionPath, relativeFile)); if (!isPathInsideDir(collectionPath, filepath)) { processed++; skippedFiles.push({ file: relativeFile, code: "OUTSIDE_COLLECTION" }); @@ -1673,13 +1742,51 @@ export async function reindexCollection( } seenPaths.add(path); + // Stat first — mtime+size fast-path (no read) + let stat: ReturnType | null = null; + try { + stat = statSync(filepath); + } catch (err) { + processed++; + skippedFiles.push({ file: relativeFile, code: fsErrorCode(err) }); + options?.onProgress?.({ file: relativeFile, current: processed, total }); + continue; + } + + if (!stat) { + processed++; + options?.onProgress?.({ file: relativeFile, current: processed, total }); + continue; + } + + const mtimeMs = stat.mtimeMs; + const size = stat.size; + + // Skip large files (>10MB) — prevents OOM + if (size > REINDEX_MAX_FILE_SIZE) { + processed++; + skippedFiles.push({ file: relativeFile, code: "FILE_TOO_LARGE" }); + options?.onProgress?.({ file: relativeFile, current: processed, total }); + continue; + } + + // Fast-path: stat matches cached sync state — skip read entirely + const cached = syncStateMap.get(path); + if (cached && cached.mtime_ms === Math.floor(mtimeMs) && cached.size === size) { + unchanged++; + processed++; + options?.onProgress?.({ file: relativeFile, current: processed, total }); + // Still need to ensure document exists (might have been deactivated externally) + // But we count as unchanged and avoid expensive read+hash+metadata sync. + // Note: metadata sync for unchanged is skipped in fast-path; if needed, disable fast-path or force re-read. + continue; + } + + // Need to read file let content: string; try { content = readFileSync(filepath, "utf-8"); } catch (err) { - // Skip files that can't be read (ETIMEDOUT on APFS compressed files, - // EAGAIN on iCloud evicted files, EACCES, etc.) instead of aborting - // the rest of the collection (#460). processed++; skippedFiles.push({ file: relativeFile, code: fsErrorCode(err) }); options?.onProgress?.({ file: relativeFile, current: processed, total }); @@ -1687,13 +1794,39 @@ export async function reindexCollection( } if (!content.trim()) { + // Empty file — if previously indexed, deactivate it (treat as removed) + const existingEmpty = findOrMigrateLegacyDocument(db, collectionName, path, livePaths); + if (existingEmpty) { + deactivateDocument(db, collectionName, path); + deleteFileSyncStateForCollection(db, collectionName, path); + syncStateMap.delete(path); + } processed++; + options?.onProgress?.({ file: relativeFile, current: processed, total }); continue; } const hash = await hashContent(content); - const title = extractTitle(content, relativeFile); + // Hash matches cached sync state but mtime differed (clock skew, backup restore) — only update mtime cache + if (cached && cached.content_hash === hash) { + // Update sync state mtime/size only + upsertFileSyncState(db, collectionName, path, mtimeMs, size, hash, cached.document_id); + unchanged++; + processed++; + // Keep content in memory for metadata sync if needed? For speed, skip metadata sync on hash-match fast-path. + // Existing behavior for hash-same was to still do metadata backfill; we preserve it by loading documentId from cache. + // However we already have content here, so do metadata backfill for hash-match case. + const existingForMeta = findOrMigrateLegacyDocument(db, collectionName, path, livePaths); + if (existingForMeta) { + const extraction = syncDocumentMetadata(db, existingForMeta.id, content, path, { onlyIfStale: true }); + if (extraction?.error) metadataErrors++; + } + options?.onProgress?.({ file: relativeFile, current: processed, total }); + continue; + } + + const title = extractTitle(content, relativeFile); const existing = findOrMigrateLegacyDocument(db, collectionName, path, livePaths); let documentId: number; @@ -1711,21 +1844,21 @@ export async function reindexCollection( } } else { insertContent(db, hash, content, now); - const stat = statSync(filepath); - updateDocument(db, existing.id, title, hash, - stat ? new Date(stat.mtime).toISOString() : now); + updateDocument(db, existing.id, title, hash, new Date(stat.mtime).toISOString()); updated++; } } else { indexed++; insertContent(db, hash, content, now); - const stat = statSync(filepath); documentId = insertDocument(db, collectionName, path, title, hash, stat ? new Date(stat.birthtime).toISOString() : now, - stat ? new Date(stat.mtime).toISOString() : now); + new Date(stat.mtime).toISOString()); } - // Unchanged content still backfills missing or stale extraction state. + // Upsert sync state after successful indexing + upsertFileSyncState(db, collectionName, path, mtimeMs, size, hash, documentId); + + // Metadata extraction const extraction = syncDocumentMetadata(db, documentId, content, path, contentChanged ? undefined : { onlyIfStale: true }); if (extraction?.error) metadataErrors++; @@ -1734,12 +1867,13 @@ export async function reindexCollection( options?.onProgress?.({ file: relativeFile, current: processed, total }); } - // Deactivate documents that no longer exist + // Deactivate documents that no longer exist + cleanup sync_state const allActive = getActiveDocumentPaths(db, collectionName); let removed = 0; for (const path of allActive) { if (!seenPaths.has(path)) { deactivateDocument(db, collectionName, path); + deleteFileSyncStateForCollection(db, collectionName, path); removed++; } } From b27634aa0e75535a97eeb4502a775397bc54ff24 Mon Sep 17 00:00:00 2001 From: Rik van Riel Date: Wed, 16 Sep 2026 14:21:59 -0400 Subject: [PATCH 15/82] fix(search): bound hydration at DB level via substr, not JS Previously, searchFTS and searchVec loaded the whole document body as content.doc as body. On collections with large transcripts this caused 20 candidates times 6 expansions times 100KB = 12MB of string copies through better-sqlite3 into V8, observed as 4.2GB heap. Now the bodies are bounded at the SQL level using substr(content.doc, 1, 262144) as body. This limits each body to 256KB. With 40 winners the max is 10MB vs unbounded before. Rerank input stays 900 tokens, about 3.6KB, per chunk. The long term fix fetches only the winning chunk via substr(doc, pos, len) for the 40 RRF winners, but this commit prevents the OOM first. (cherry picked from commit a75b659d0ec71b56e4c7280c59737ed3f578e7a8) --- src/store.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/store.ts b/src/store.ts index bb76647b4..c724f0f11 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4324,7 +4324,7 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle 'qmd://' || d.collection || '/' || d.path as filepath, d.collection || '/' || d.path as display_path, d.title, - content.doc as body, + substr(content.doc, 1, 262144) as body, d.hash, fm.bm25_score, dm.metadata_json @@ -4530,7 +4530,7 @@ export async function searchVec(db: Database, query: string, model: string, limi 'qmd://' || d.collection || '/' || d.path as filepath, d.collection || '/' || d.path as display_path, d.title, - content.doc as body, + substr(content.doc, 1, 262144) as body, dm.metadata_json FROM content_vectors cv JOIN documents d ON d.hash = cv.hash AND d.active = 1 From a3791ceb606164bd5cedb8aaa3d4969e05110bc8 Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Sun, 13 Sep 2026 15:30:43 -0700 Subject: [PATCH 16/82] feat(metadata): add text and type operators to the filter language Extend the filter AST with three text operators, a type test, and an optional case-folding flag, so a filter can reach values by substring and reach one side of a key whose documents disagree on type. - `contains`, `prefix`, and `suffix` take a non-empty string operand and match string values only. `contains` compiles to `instr()`. `prefix` and `suffix` compare UTF-8 bytes with the operand's byte length bound from JavaScript, because SQLite's text `length()` and `substr()` stop at an embedded NUL and the stored string domain admits one. - `type` takes `string`, `number`, or `boolean` and matches the stored type of a key's values. - `caseInsensitive: true` is accepted on any condition whose operand is a string or an array of strings. Both sides fold ASCII letters (SQLite's `lower()` and the same range in JavaScript), so the two agree exactly. Non-ASCII letters compare exactly. Each compiled condition now names how its correlated subquery reaches a document's rows. The covering indexes are (key, value, document_id), so an exact equality (`eq`, `in`, `all`) is one probe by value. Every other condition (ordered comparisons, `ne` and `nin` presence, `exists`, `type`, the text operators, and folded equalities, since `lower()` is opaque to the index) seeks the document's own rows through the primary key: a unary `+` on the key term hides it from the planner, which otherwise walks the key's whole value range once per document when `sqlite_stat1` is absent. On 4,000 documents that plan made one `priority > 10` count 376 ms and a `prefix` count 946 ms; both are under 5 ms with the seek, on Node and Bun, with or without statistics. A test asserts the plan for every operator in both states. Validation reports the failing JSON path as before. The README filter table, the MCP `query` tool description, the skill, and the changelog entry name the new operators. Bind vector candidate IDs as one JSON list during document lookup. A valid near-ceiling filter otherwise overflows Node's SQL variable limit when combined with candidate IDs, in both exact scans and the global fallback. A model-free regression crosses the 20,000-vector boundary and verifies that nonmatching documents sharing a content hash remain excluded. The plan guard accepts SQLite 3.43's USING INDEX label for the same point seek, retaining the check that rejects broad per-document key scans. Assisted-by: Claude Fable 5.1 via Pi --- CHANGELOG.md | 2 +- README.md | 8 +- skills/qmd/SKILL.md | 2 +- src/index.ts | 6 + src/mcp/server.ts | 4 +- src/metadata-filter.ts | 253 +++++++++++++++++++++++++++-------- src/metadata.ts | 3 + src/store.ts | 9 +- test/metadata-filter.test.ts | 172 ++++++++++++++++++++++++ test/metadata-search.test.ts | 33 ++++- 10 files changed, 426 insertions(+), 66 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1ad02c921..2524cf779 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,7 +5,7 @@ ### Added - Added Oxlint lint fence. -- Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter `, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator` (conditions are `{ field, operator, value }`, where `field` names the metadata key): `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, and `exists` presence. Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes). +- Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter `, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator`: `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, `contains`/`prefix`/`suffix` text matching, `type` for the stored type of a key's values, and `exists` presence, with an optional `caseInsensitive` flag on any condition whose value is a string (ASCII folding). Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes). ### Fixed diff --git a/README.md b/README.md index 40ff97b10..677dbfe58 100644 --- a/README.md +++ b/README.md @@ -957,6 +957,7 @@ qmd query "dependency injection" --filter '{ { "field": "priority", "operator": "gte", "value": 3 }, { "field": "reviewed", "operator": "eq", "value": true } ] }, + { "field": "topics", "operator": "prefix", "value": "sql", "caseInsensitive": true }, { "operator": "not", "operand": { "field": "audience", "operator": "eq", "value": "internal" } } ] }' @@ -968,13 +969,18 @@ qmd query "dependency injection" --filter '{ | Negation | `{ "operator": "not", "operand": {…} }` | | Comparison | `{ "field", "operator": "eq" \| "ne" \| "gt" \| "gte" \| "lt" \| "lte", "value" }` | | Membership | `{ "field", "operator": "in" \| "nin" \| "all", "value": […] }` | +| Text | `{ "field", "operator": "contains" \| "prefix" \| "suffix", "value": "…" }` | +| Type | `{ "field", "operator": "type", "value": "string" \| "number" \| "boolean" }` | | Presence | `{ "field", "operator": "exists", "value": true \| false }` | +Any condition whose value is a string or an array of strings may add `"caseInsensitive": true`. + Semantics: -- Matching is typed and exact — no string/number/boolean coercion, and a type mismatch never matches (including `ne` and `nin`). +- Matching is typed and exact — no string/number/boolean coercion, and a type mismatch never matches (including `ne` and `nin`). Text operators match string values only. `type` matches the stored type of a key's values, which is how a filter reaches one side of a key whose documents disagree on type. - Array-valued metadata is a set: a condition matches when any element satisfies it, `all` requires every filter value to be present. - Missing keys do not match `ne`/`nin`; combine with `{ "operator": "exists", "value": false }` in an `or` group to include them. +- Matching is case-sensitive unless a condition sets `caseInsensitive`, which folds ASCII letters on both sides. Non-ASCII letters compare exactly. - Multiple conditions require an explicit `and` group — there is no implicit AND, and no `$`-prefixed shorthand. Guarantees and limits: diff --git a/skills/qmd/SKILL.md b/skills/qmd/SKILL.md index 68dbbbb80..08b1a764a 100644 --- a/skills/qmd/SKILL.md +++ b/skills/qmd/SKILL.md @@ -208,7 +208,7 @@ qmd search "authentication" --filter '{"field":"status","operator":"eq","value": qmd query "dependency injection" --filter '{"operator":"and","operands":[{"field":"topics","operator":"all","value":["typescript"]},{"field":"status","operator":"nin","value":["draft","archived"]}]}' ``` -Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` takes one `operand`, and conditions take `field` + `value` with operators `eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), or `exists` (presence). Matching is typed and exact; missing keys do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to include them). The MCP `query` tool accepts the same AST as a `filter` object. JSON output includes each result's `metadata`. +Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` takes one `operand`, and conditions take `field` + `value` with operators `eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), `contains`/`prefix`/`suffix` (text), `type` (the value is a `string`, `number`, or `boolean`), or `exists` (presence). Matching is typed and exact; missing keys do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to include them). Conditions with a string value may add `"caseInsensitive": true`, which folds ASCII letters. The MCP `query` tool accepts the same AST as a `filter` object. JSON output includes each result's `metadata`. ## MCP Tool: `query` diff --git a/src/index.ts b/src/index.ts index 223c10d9d..5ac03bb34 100644 --- a/src/index.ts +++ b/src/index.ts @@ -81,6 +81,9 @@ import { type MetadataFilterGroup, type MetadataFilterNegation, type MetadataCondition, + type MetadataPredicate, + type MetadataPredicateGroup, + type MetadataPredicateNegation, } from "./metadata-filter.js"; import { setConfigSource, @@ -133,6 +136,9 @@ export type { MetadataFilterGroup, MetadataFilterNegation, MetadataCondition, + MetadataPredicate, + MetadataPredicateGroup, + MetadataPredicateNegation, }; export { parseMetadataFilter, MetadataFilterError }; diff --git a/src/mcp/server.ts b/src/mcp/server.ts index c92efc110..e0e05a31e 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -349,8 +349,10 @@ Intent-aware lex (C++ performance, not sports): "Metadata filter (recursive JSON AST). Every returned result satisfies it. " + "Nodes are operator-discriminated: logical groups {operator:'and'|'or', operands:[...]}, " + "negation {operator:'not', operand:{...}}, and conditions {field, operator, value} with " + - "operators eq/ne/gt/gte/lt/lte (comparison), in/nin/all (membership), exists (presence). " + + "operators eq/ne/gt/gte/lt/lte (comparison), in/nin/all (membership), contains/prefix/suffix (text), " + + "type (value is 'string'|'number'|'boolean'), exists (presence). " + "Values are typed exactly (no coercion); missing keys do not match ne/nin. " + + "Conditions with a string value may add caseInsensitive:true (ASCII folding). " + "Example: {\"operator\":\"and\",\"operands\":[{\"field\":\"topics\",\"operator\":\"all\",\"value\":[\"typescript\"]}," + "{\"field\":\"status\",\"operator\":\"ne\",\"value\":\"draft\"}]}" ), diff --git a/src/metadata-filter.ts b/src/metadata-filter.ts index 5456aeff8..8c22a44d6 100644 --- a/src/metadata-filter.ts +++ b/src/metadata-filter.ts @@ -1,49 +1,67 @@ /** - * QMD Metadata Filter - Recursive filter AST, strict runtime validation, and - * parameterized SQL compilation. + * QMD Metadata Filter - Recursive predicate AST, strict runtime validation, + * and parameterized SQL compilation. * - * The filter has one canonical, `operator`-discriminated recursive shape shared - * by every public search surface (CLI, SDK, MCP, HTTP): + * The predicate has one canonical, `operator`-discriminated recursive shape + * shared by every public search surface (CLI, SDK, MCP, HTTP): * * { "operator": "and", "operands": [ ... ] } * { "operator": "not", "operand": { ... } } * { "field": "status", "operator": "eq", "value": "published" } + * { "field": "topics", "operator": "prefix", "value": "sql", "caseInsensitive": true } * - * A condition is a predicate over one field of the record under evaluation: - * `field` names it, `operator` says how to compare, `value` is the operand. - * For a document, the fields are its metadata keys. + * A condition tests one field of the record under evaluation, named by `field`, + * against `value` using `operator`. For a document the fields are its metadata + * keys. Comparison, membership, and text operators are typed by their operand: + * a string operand only ever compares against string values, a number against + * numbers, a boolean against booleans, and a mismatch never matches. * * Compilation emits correlated EXISTS/NOT EXISTS subqueries over - * `document_metadata_values` with every user value bound as a parameter — - * metadata keys and values are data, never SQL. + * `document_metadata_values` with every user value bound as a parameter. + * Metadata keys and values are data, never SQL. */ -import type { MetadataScalar, MetadataScalarArray } from "./metadata.js"; +import type { MetadataScalar, MetadataScalarArray, MetadataValueType } from "./metadata.js"; import { METADATA_LIMITS } from "./metadata.js"; // ============================================================================= // Public types // ============================================================================= -export type MetadataFilter = - | MetadataFilterGroup - | MetadataFilterNegation - | MetadataCondition; +/** + * The recursive grammar: a condition, or `and`/`or`/`not` over predicates. + * `Condition` is the set of conditions the record under evaluation admits. + */ +export type MetadataPredicate = + | Condition + | MetadataPredicateGroup + | MetadataPredicateNegation; -export interface MetadataFilterGroup { +export interface MetadataPredicateGroup { operator: "and" | "or"; - operands: readonly MetadataFilter[]; + operands: readonly MetadataPredicate[]; } -export interface MetadataFilterNegation { +export interface MetadataPredicateNegation { operator: "not"; - operand: MetadataFilter; + operand: MetadataPredicate; } +/** A predicate over a document, whose fields are its metadata keys. */ +export type MetadataFilter = MetadataPredicate; +export type MetadataFilterGroup = MetadataPredicateGroup; +export type MetadataFilterNegation = MetadataPredicateNegation; + +/** + * One condition. `caseInsensitive` folds ASCII letters on both sides and is + * accepted only where the operand is a string or an array of strings. + */ export type MetadataCondition = - | { field: string; operator: "eq" | "ne"; value: MetadataScalar } - | { field: string; operator: "gt" | "gte" | "lt" | "lte"; value: string | number } - | { field: string; operator: "in" | "nin" | "all"; value: MetadataScalarArray } + | { field: string; operator: "eq" | "ne"; value: MetadataScalar; caseInsensitive?: boolean } + | { field: string; operator: "gt" | "gte" | "lt" | "lte"; value: string | number; caseInsensitive?: boolean } + | { field: string; operator: "in" | "nin" | "all"; value: MetadataScalarArray; caseInsensitive?: boolean } + | { field: string; operator: "contains" | "prefix" | "suffix"; value: string; caseInsensitive?: boolean } + | { field: string; operator: "type"; value: MetadataValueType } | { field: string; operator: "exists"; value: boolean }; export interface CompiledMetadataFilter { @@ -80,8 +98,10 @@ const GROUP_OPERATORS = new Set(["and", "or"]); const COMPARISON_OPERATORS = new Set(["eq", "ne", "gt", "gte", "lt", "lte"]); const ORDERED_OPERATORS = new Set(["gt", "gte", "lt", "lte"]); const MEMBERSHIP_OPERATORS = new Set(["in", "nin", "all"]); -const CONDITION_OPERATORS = new Set([...COMPARISON_OPERATORS, ...MEMBERSHIP_OPERATORS, "exists"]); +const TEXT_OPERATORS = new Set(["contains", "prefix", "suffix"]); +const CONDITION_OPERATORS = new Set([...COMPARISON_OPERATORS, ...MEMBERSHIP_OPERATORS, ...TEXT_OPERATORS, "type", "exists"]); const ALL_OPERATORS = [...GROUP_OPERATORS, "not", ...CONDITION_OPERATORS]; +const VALUE_TYPES: readonly MetadataValueType[] = ["string", "number", "boolean"]; // ============================================================================= // Validation @@ -89,6 +109,9 @@ const ALL_OPERATORS = [...GROUP_OPERATORS, "not", ...CONDITION_OPERATORS]; type FilterParseState = { nodes: number }; +/** Conditions whose operand can be a string, and so can fold case. */ +type CaseFoldableCondition = Exclude; + /** * Strictly validate an untrusted value as a MetadataFilter. * Rejects unknown operators, unknown properties, operator-incompatible values, @@ -179,7 +202,7 @@ function parseFilterNegation( } function parseFilterCondition(node: Record, operator: string, path: string): MetadataCondition { - rejectUnknownProperties(node, ["field", "operator", "value"], path); + rejectUnknownProperties(node, ["field", "operator", "value", "caseInsensitive"], path); const field = node["field"]; if (typeof field !== "string" || field.length === 0) { @@ -194,19 +217,39 @@ function parseFilterCondition(node: Record, operator: string, p } const value = node["value"]; + const caseInsensitive = node["caseInsensitive"]; + if (caseInsensitive !== undefined && typeof caseInsensitive !== "boolean") { + throw new MetadataFilterError(`${path}.caseInsensitive`, "'caseInsensitive' must be a boolean"); + } + if (operator === "exists") { if (typeof value !== "boolean") { throw new MetadataFilterError(`${path}.value`, "'exists' requires a boolean value"); } + rejectCaseInsensitive(caseInsensitive, operator, path); return { field, operator, value }; } + if (operator === "type") { + const valueType = VALUE_TYPES.find(candidate => candidate === value); + if (!valueType) { + throw new MetadataFilterError(`${path}.value`, `'type' requires one of: ${VALUE_TYPES.join(", ")}`); + } + rejectCaseInsensitive(caseInsensitive, operator, path); + return { field, operator, value: valueType }; + } + + if (TEXT_OPERATORS.has(operator)) { + const text = parseScalarValue(value, `${path}.value`); + if (typeof text !== "string" || text.length === 0) { + throw new MetadataFilterError(`${path}.value`, `'${operator}' requires a non-empty string value`); + } + return withCaseFolding({ field, operator: operator as "contains" | "prefix" | "suffix", value: text }, caseInsensitive, path); + } + if (MEMBERSHIP_OPERATORS.has(operator)) { - return { - field, - operator: operator as "in" | "nin" | "all", - value: parseMembershipValues(value, operator, path), - }; + const values = parseMembershipValues(value, operator, path); + return withCaseFolding({ field, operator: operator as "in" | "nin" | "all", value: values }, caseInsensitive, path); } // Comparison operators: eq, ne, gt, gte, lt, lte. @@ -214,7 +257,26 @@ function parseFilterCondition(node: Record, operator: string, p if (ORDERED_OPERATORS.has(operator) && typeof scalar === "boolean") { throw new MetadataFilterError(`${path}.value`, `'${operator}' requires a string or number value`); } - return { field, operator, value: scalar } as MetadataCondition; + return withCaseFolding({ field, operator, value: scalar } as CaseFoldableCondition, caseInsensitive, path); +} + +/** Attach `caseInsensitive` when given, which only string operands accept. */ +function withCaseFolding(condition: CaseFoldableCondition, caseInsensitive: unknown, path: string): MetadataCondition { + if (caseInsensitive === undefined) return condition; + + const operand: unknown = condition.value; + const isText = typeof operand === "string"; + const isTextArray = Array.isArray(operand) && operand.every(element => typeof element === "string"); + if (!isText && !isTextArray) { + throw new MetadataFilterError(`${path}.caseInsensitive`, "'caseInsensitive' applies to string values only"); + } + + return { ...condition, caseInsensitive: caseInsensitive === true }; +} + +function rejectCaseInsensitive(caseInsensitive: unknown, operator: string, path: string): void { + if (caseInsensitive === undefined) return; + throw new MetadataFilterError(`${path}.caseInsensitive`, `'caseInsensitive' does not apply to '${operator}'`); } function parseMembershipValues(value: unknown, operator: string, path: string): MetadataScalarArray { @@ -299,8 +361,12 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin case "exists": params.push(filter.field); return filter.value - ? buildValueExistsSql(alias, "mv.key = ?") - : `NOT ${buildValueExistsSql(alias, "mv.key = ?")}`; + ? buildValueExistsSql(alias, "byDocument") + : `NOT ${buildValueExistsSql(alias, "byDocument")}`; + + case "type": + params.push(filter.field, filter.value); + return buildValueExistsSql(alias, "byDocument", "mv.value_type = ?"); case "eq": case "gt": @@ -308,10 +374,12 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin case "lt": case "lte": { const sqlOperator = { eq: "=", gt: ">", gte: ">=", lt: "<", lte: "<=" }[filter.operator]; - params.push(filter.field, bindScalar(filter.value)); + const columnSql = buildValueColumnSql(filter.value, filter.caseInsensitive); + params.push(filter.field, bindOperand(filter.value, filter.caseInsensitive)); return buildValueExistsSql( alias, - `mv.key = ? AND mv.value_type = '${valueTypeOf(filter.value)}' AND mv.${valueColumnOf(filter.value)} ${sqlOperator} ?`, + filter.operator === "eq" ? equalitySeekOf(filter.caseInsensitive) : "byDocument", + `mv.value_type = '${valueTypeOf(filter.value)}' AND ${columnSql} ${sqlOperator} ?`, ); } @@ -319,61 +387,132 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin // Key must have at least one same-type value, and no same-type value // may equal the operand. Missing keys and type mismatches do not match. const valueType = valueTypeOf(filter.value); + const columnSql = buildValueColumnSql(filter.value, filter.caseInsensitive); params.push(filter.field); - const presentSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}'`); - params.push(filter.field, bindScalar(filter.value)); - const equalSql = buildValueExistsSql( - alias, - `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${valueColumnOf(filter.value)} = ?`, - ); + const presentSql = buildValueExistsSql(alias, "byDocument", `mv.value_type = '${valueType}'`); + params.push(filter.field, bindOperand(filter.value, filter.caseInsensitive)); + const equalSql = buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} = ?`); return `(${presentSql} AND NOT ${equalSql})`; } case "in": case "nin": { const valueType = valueTypeOf(filter.value[0]!); - const column = valueColumnOf(filter.value[0]!); + const columnSql = buildValueColumnSql(filter.value[0]!, filter.caseInsensitive); const placeholders = filter.value.map(() => "?").join(", "); + const operands = filter.value.map(element => bindOperand(element, filter.caseInsensitive)); if (filter.operator === "in") { - params.push(filter.field, ...filter.value.map(bindScalar)); - return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} IN (${placeholders})`); + params.push(filter.field, ...operands); + return buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} IN (${placeholders})`); } params.push(filter.field); - const presentSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}'`); - params.push(filter.field, ...filter.value.map(bindScalar)); - const memberSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} IN (${placeholders})`); + const presentSql = buildValueExistsSql(alias, "byDocument", `mv.value_type = '${valueType}'`); + params.push(filter.field, ...operands); + const memberSql = buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} IN (${placeholders})`); return `(${presentSql} AND NOT ${memberSql})`; } case "all": { const valueType = valueTypeOf(filter.value[0]!); - const column = valueColumnOf(filter.value[0]!); + const columnSql = buildValueColumnSql(filter.value[0]!, filter.caseInsensitive); const memberSqls = filter.value.map(element => { - params.push(filter.field, bindScalar(element)); - return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} = ?`); + params.push(filter.field, bindOperand(element, filter.caseInsensitive)); + return buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} = ?`); }); return `(${memberSqls.join(" AND ")})`; } + + case "contains": + case "prefix": + case "suffix": { + params.push(filter.field); + const textSql = compileTextTestSql(filter.operator, filter.value, filter.caseInsensitive, params); + return buildValueExistsSql(alias, "byDocument", `mv.value_type = 'string' AND ${textSql}`); + } + } +} + +/** + * How a correlated subquery reaches one document's rows for a key. The + * covering indexes are (key, , document_id), so a condition that pins + * the value column with an equality is a single probe by value. Any other + * condition would walk the key's whole value range once per document (the + * stats-free planner assumes a range is narrow), so it seeks the document's + * own rows through the primary key instead. + */ +type RowSeek = "byValue" | "byDocument"; + +/** An equality pins the value column unless it is folded, `lower()` is opaque to the index. */ +function equalitySeekOf(caseInsensitive: boolean | undefined): RowSeek { + return caseInsensitive ? "byDocument" : "byValue"; +} + +/** + * `EXISTS` over the document's rows for the key, whose parameter the caller + * binds before `conditionSql`'s. The unary `+` on the key term hides it from + * the planner, which leaves `document_id` as the only indexable term and + * makes the seek independent of `sqlite_stat1`. + */ +function buildValueExistsSql(alias: string, seek: RowSeek, conditionSql?: string): string { + const keyTermSql = seek === "byValue" ? "mv.key = ?" : "+mv.key = ?"; + const whereSql = conditionSql ? `${keyTermSql} AND ${conditionSql}` : keyTermSql; + return `EXISTS (SELECT 1 FROM document_metadata_values mv WHERE mv.document_id = ${alias}.id AND ${whereSql})`; +} + +/** + * Substring tests over the string column. Prefix and suffix compare UTF-8 + * bytes, with the operand's byte length bound from JavaScript: SQLite's text + * `length()` and `substr()` stop at an embedded NUL, and blobs do not. UTF-8 + * is self-synchronizing, so a byte-prefix (or byte-suffix) of a whole operand + * is exactly a character-prefix (or -suffix). + */ +function compileTextTestSql( + operator: "contains" | "prefix" | "suffix", + text: string, + caseInsensitive: boolean | undefined, + params: (string | number)[], +): string { + const columnSql = buildValueColumnSql(text, caseInsensitive); + const operand = caseInsensitive ? foldAsciiCase(text) : text; + + if (operator === "contains") { + params.push(operand); + return `instr(${columnSql}, ?) > 0`; } + + params.push(utf8ByteLengthOf(operand), operand); + return operator === "prefix" + ? `substr(CAST(${columnSql} AS BLOB), 1, ?) = CAST(? AS BLOB)` + : `substr(CAST(${columnSql} AS BLOB), -?) = CAST(? AS BLOB)`; } -function buildValueExistsSql(alias: string, conditionSql: string): string { - return `EXISTS (SELECT 1 FROM document_metadata_values mv WHERE mv.document_id = ${alias}.id AND ${conditionSql})`; +const utf8Encoder = new TextEncoder(); + +function utf8ByteLengthOf(text: string): number { + return utf8Encoder.encode(text).byteLength; } -function valueTypeOf(scalar: MetadataScalar): "string" | "number" | "boolean" { - return typeof scalar as "string" | "number" | "boolean"; +function valueTypeOf(scalar: MetadataScalar): MetadataValueType { + return typeof scalar as MetadataValueType; } -function valueColumnOf(scalar: MetadataScalar): "text_value" | "number_value" | "boolean_value" { - if (typeof scalar === "string") return "text_value"; - if (typeof scalar === "number") return "number_value"; - return "boolean_value"; +/** The typed column an operand compares against, folded when the condition ignores case. */ +function buildValueColumnSql(scalar: MetadataScalar, caseInsensitive: boolean | undefined): string { + if (typeof scalar === "number") return "mv.number_value"; + if (typeof scalar === "boolean") return "mv.boolean_value"; + return caseInsensitive ? "lower(mv.text_value)" : "mv.text_value"; } -function bindScalar(scalar: MetadataScalar): string | number { +/** Booleans bind as 0/1. Case-insensitive strings bind folded the same way SQLite's `lower()` folds the column. */ +function bindOperand(scalar: MetadataScalar, caseInsensitive: boolean | undefined): string | number { if (typeof scalar === "boolean") return scalar ? 1 : 0; + if (typeof scalar === "string" && caseInsensitive) return foldAsciiCase(scalar); return scalar; } + +/** SQLite's built-in `lower()` folds ASCII letters only, so the operand is folded over the same range. */ +function foldAsciiCase(text: string): string { + return text.replace(/[A-Z]/g, letter => letter.toLowerCase()); +} diff --git a/src/metadata.ts b/src/metadata.ts index 602bebfd1..5ddc9082e 100644 --- a/src/metadata.ts +++ b/src/metadata.ts @@ -28,6 +28,9 @@ import YAML from "yaml"; export type MetadataScalar = string | number | boolean; +/** The stored type of one metadata value. Arrays are homogeneous, so a key holds one type per document. */ +export type MetadataValueType = "string" | "number" | "boolean"; + export type MetadataScalarArray = | readonly string[] | readonly number[] diff --git a/src/store.ts b/src/store.ts index 3a45bbcb9..936d0f61c 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4311,8 +4311,9 @@ export async function searchVec(db: Database, query: string, model: string, limi const hashSeqs = vecResults.map(r => r.hash_seq); const distanceMap = new Map(vecResults.map(r => [r.hash_seq, r.distance])); - // Build query for document lookup - const placeholders = hashSeqs.map(() => '?').join(','); + // One binding for candidate IDs leaves room for a valid near-ceiling + // metadata filter under Node's 32,766-variable limit. The same lookup + // serves exact scans and the capped global fallback. let docSql = ` SELECT cv.hash || '_' || cv.seq as hash_seq, @@ -4327,9 +4328,9 @@ export async function searchVec(db: Database, query: string, model: string, limi JOIN documents d ON d.hash = cv.hash AND d.active = 1 JOIN content ON content.hash = d.hash LEFT JOIN document_metadata dm ON dm.document_id = d.id - WHERE cv.hash || '_' || cv.seq IN (${placeholders}) + WHERE cv.hash || '_' || cv.seq IN (SELECT value FROM json_each(?)) `; - const params: (string | number)[] = [...hashSeqs]; + const params: (string | number)[] = [JSON.stringify(hashSeqs)]; if (collectionFilter) { docSql += ` AND d.collection = ?`; diff --git a/test/metadata-filter.test.ts b/test/metadata-filter.test.ts index 0d1c883fd..7510fd2ee 100644 --- a/test/metadata-filter.test.ts +++ b/test/metadata-filter.test.ts @@ -33,12 +33,43 @@ describe("parseMetadataFilter", () => { { field: "topics", operator: "nin", value: [1, 2] }, { field: "flags", operator: "all", value: [true, false] }, { field: "status", operator: "exists", value: false }, + { field: "topics", operator: "contains", value: "vec" }, + { field: "topics", operator: "prefix", value: "sql" }, + { field: "owner", operator: "suffix", value: "-team" }, + { field: "priority", operator: "type", value: "number" }, ]; for (const condition of conditions) { expect(parseMetadataFilter(condition)).toEqual(condition); } }); + test("accepts caseInsensitive on string operands only", () => { + const accepted: unknown[] = [ + { field: "status", operator: "eq", value: "Published", caseInsensitive: true }, + { field: "status", operator: "ne", value: "Draft", caseInsensitive: false }, + { field: "name", operator: "gt", value: "M", caseInsensitive: true }, + { field: "topics", operator: "in", value: ["SQL", "TypeScript"], caseInsensitive: true }, + { field: "topics", operator: "all", value: ["SQL"], caseInsensitive: true }, + { field: "topics", operator: "contains", value: "SQL", caseInsensitive: true }, + { field: "topics", operator: "prefix", value: "SQL", caseInsensitive: true }, + { field: "topics", operator: "suffix", value: "SQL", caseInsensitive: true }, + ]; + for (const condition of accepted) { + expect(parseMetadataFilter(condition)).toEqual(condition); + } + + const rejected: [unknown, RegExp][] = [ + [{ field: "priority", operator: "eq", value: 3, caseInsensitive: true }, /at \$\.caseInsensitive:.*string values only/], + [{ field: "reviewed", operator: "in", value: [true], caseInsensitive: true }, /string values only/], + [{ field: "reviewed", operator: "exists", value: true, caseInsensitive: true }, /does not apply to 'exists'/], + [{ field: "priority", operator: "type", value: "number", caseInsensitive: true }, /does not apply to 'type'/], + [{ field: "status", operator: "eq", value: "x", caseInsensitive: "yes" }, /must be a boolean/], + ]; + for (const [input, expected] of rejected) { + expect(() => parseMetadataFilter(input)).toThrow(expected); + } + }); + test("accepts nested groups and negation", () => { const filter = { operator: "and", @@ -83,6 +114,11 @@ describe("parseMetadataFilter", () => { [{ operator: "in", field: "a", value: [] }, /non-empty array/], [{ operator: "in", field: "a", value: [1, "two"] }, /homogeneous array/], [{ operator: "exists", field: "a", value: "yes" }, /boolean value/], + [{ operator: "contains", field: "a", value: "" }, /at \$\.value:.*'contains' requires a non-empty string/], + [{ operator: "prefix", field: "a", value: 3 }, /'prefix' requires a non-empty string/], + [{ operator: "suffix", field: "a", value: ["x"] }, /string, number, or boolean/], + [{ operator: "type", field: "a", value: "integer" }, /at \$\.value:.*'type' requires one of: string, number, boolean/], + [{ operator: "type", field: "a", value: 1 }, /'type' requires one of/], ]; for (const [input, expected] of cases) { @@ -262,6 +298,93 @@ describe("compileMetadataFilter semantics", () => { expect(matchPaths(rangeFilter)).toEqual(["narrow.md", "wide.md"]); }); + test("text operators match any string element and never a number or boolean", () => { + insertDoc("sqlite.md", { topics: ["sqlite", "search"] }); + insertDoc("vec.md", { topics: ["sqlite-vec", "embeddings"] }); + insertDoc("pg.md", { topics: "postgres" }); + insertDoc("num.md", { topics: 42 }); + insertDoc("bool.md", { topics: true }); + + expect(matchPaths({ field: "topics", operator: "prefix", value: "sql" })).toEqual(["sqlite.md", "vec.md"]); + expect(matchPaths({ field: "topics", operator: "suffix", value: "vec" })).toEqual(["vec.md"]); + expect(matchPaths({ field: "topics", operator: "contains", value: "arch" })).toEqual(["sqlite.md"]); + expect(matchPaths({ field: "topics", operator: "contains", value: "4" })).toEqual([]); + expect(matchPaths({ field: "topics", operator: "prefix", value: "tru" })).toEqual([]); + }); + + test("prefix and suffix compare whole characters and never overrun the value", () => { + insertDoc("exact.md", { code: "abc" }); + insertDoc("emoji.md", { code: "\u{1F600}abc" }); + + // The operand equal to the value is both its prefix and its suffix. + expect(matchPaths({ field: "code", operator: "prefix", value: "abc" })).toEqual(["exact.md"]); + expect(matchPaths({ field: "code", operator: "suffix", value: "abc" })).toEqual(["emoji.md", "exact.md"]); + // An operand longer than the value cannot match either end. + expect(matchPaths({ field: "code", operator: "prefix", value: "abcd" })).toEqual([]); + expect(matchPaths({ field: "code", operator: "suffix", value: "zabc" })).toEqual([]); + // Astral characters count as one character on both sides. + expect(matchPaths({ field: "code", operator: "prefix", value: "\u{1F600}a" })).toEqual(["emoji.md"]); + expect(matchPaths({ field: "code", operator: "suffix", value: "\u{1F600}abc" })).toEqual(["emoji.md"]); + }); + + test("text operators see the whole value past an embedded NUL", () => { + insertDoc("nul.md", { code: "abc\u0000XYZ" }); + insertDoc("plain.md", { code: "abc" }); + + expect(matchPaths({ field: "code", operator: "suffix", value: "XYZ" })).toEqual(["nul.md"]); + expect(matchPaths({ field: "code", operator: "suffix", value: "abc" })).toEqual(["plain.md"]); + expect(matchPaths({ field: "code", operator: "prefix", value: "abc\u0000XYZ" })).toEqual(["nul.md"]); + expect(matchPaths({ field: "code", operator: "suffix", value: "abc\u0000XYZ" })).toEqual(["nul.md"]); + expect(matchPaths({ field: "code", operator: "prefix", value: "abc\u0000" })).toEqual(["nul.md"]); + expect(matchPaths({ field: "code", operator: "contains", value: "\u0000X" })).toEqual(["nul.md"]); + expect(matchPaths({ field: "code", operator: "suffix", value: "xyz", caseInsensitive: true })).toEqual(["nul.md"]); + expect(matchPaths({ field: "code", operator: "eq", value: "abc\u0000XYZ" })).toEqual(["nul.md"]); + }); + + test("type matches the stored type of a key's values", () => { + insertDoc("num.md", { priority: 3 }); + insertDoc("nums.md", { priority: [1, 2] }); + insertDoc("str.md", { priority: "high" }); + insertDoc("bool.md", { priority: true }); + insertDoc("none.md", { other: 1 }); + + expect(matchPaths({ field: "priority", operator: "type", value: "number" })).toEqual(["num.md", "nums.md"]); + expect(matchPaths({ field: "priority", operator: "type", value: "string" })).toEqual(["str.md"]); + expect(matchPaths({ field: "priority", operator: "type", value: "boolean" })).toEqual(["bool.md"]); + expect(matchPaths({ + operator: "not", + operand: { field: "priority", operator: "type", value: "string" }, + })).toEqual(["bool.md", "none.md", "num.md", "nums.md"]); + }); + + test("caseInsensitive folds ASCII letters on both sides for every string operator", () => { + insertDoc("upper.md", { status: "PUBLISHED", topics: ["PostgreSQL", "MySQL"] }); + insertDoc("lower.md", { status: "published", topics: ["sqlite"] }); + insertDoc("draft.md", { status: "Draft", topics: ["Search"] }); + + expect(matchPaths({ field: "status", operator: "eq", value: "Published" })).toEqual([]); + expect(matchPaths({ field: "status", operator: "eq", value: "Published", caseInsensitive: true })).toEqual(["lower.md", "upper.md"]); + expect(matchPaths({ field: "status", operator: "ne", value: "published", caseInsensitive: true })).toEqual(["draft.md"]); + expect(matchPaths({ field: "status", operator: "in", value: ["DRAFT"], caseInsensitive: true })).toEqual(["draft.md"]); + expect(matchPaths({ field: "status", operator: "nin", value: ["DRAFT"], caseInsensitive: true })).toEqual(["lower.md", "upper.md"]); + expect(matchPaths({ field: "topics", operator: "all", value: ["postgresql", "mysql"], caseInsensitive: true })).toEqual(["upper.md"]); + expect(matchPaths({ field: "topics", operator: "contains", value: "sql", caseInsensitive: true })).toEqual(["lower.md", "upper.md"]); + expect(matchPaths({ field: "topics", operator: "prefix", value: "postgres", caseInsensitive: true })).toEqual(["upper.md"]); + expect(matchPaths({ field: "topics", operator: "suffix", value: "SQL", caseInsensitive: true })).toEqual(["upper.md"]); + // Lexical comparison folds too: "Draft" sorts before "published" once lowered. + expect(matchPaths({ field: "status", operator: "lt", value: "M", caseInsensitive: true })).toEqual(["draft.md"]); + }); + + test("caseInsensitive leaves non-ASCII letters exact", () => { + insertDoc("upper.md", { city: "\u00C9VORA" }); + insertDoc("lower.md", { city: "\u00E9vora" }); + + // The ASCII part folds and the accented initial does not, so each + // spelling matches itself only. + expect(matchPaths({ field: "city", operator: "eq", value: "\u00C9vora", caseInsensitive: true })).toEqual(["upper.md"]); + expect(matchPaths({ field: "city", operator: "eq", value: "\u00E9VORA", caseInsensitive: true })).toEqual(["lower.md"]); + }); + test("boolean values round-trip through membership operators", () => { insertDoc("flagged.md", { reviewed: true }); insertDoc("unflagged.md", { reviewed: false }); @@ -283,12 +406,59 @@ describe("compileMetadataFilter semantics", () => { expect((db.prepare(`SELECT COUNT(*) as c FROM documents`).get() as { c: number }).c).toBe(1); }); + test("every condition seeks by value or by document, with and without statistics", () => { + // The covering indexes are (key, , document_id). An equality on the + // value column is one probe by value. Everything else must seek the + // document's rows through the primary key, or the planner walks the key's + // whole value range once per document. `ne` and `nin` pair one of each. + // ANALYZE must not change any of it (below about 32 rows, statistics + // make a full scan of the tiny index the cheaper plan, correctly). + for (let index = 0; index < 64; index++) { + insertDoc(`d${index}.md`, { status: index % 2 ? "published" : "draft", priority: index, topics: [`t${index}`] }); + } + + // SQLite 3.43 labels the same point probe USING INDEX, without COVERING. + const byValue = "INDEX idx_metadata_"; + const byDocument = "INDEX sqlite_autoindex_document_metadata_values_1 (document_id=?)"; + const expectations: [MetadataFilter, string][] = [ + [{ field: "status", operator: "eq", value: "published" }, byValue], + [{ field: "topics", operator: "in", value: ["t1", "t2"] }, byValue], + [{ field: "topics", operator: "all", value: ["t1"] }, byValue], + [{ field: "status", operator: "eq", value: "PUBLISHED", caseInsensitive: true }, byDocument], + [{ field: "priority", operator: "gt", value: 3 }, byDocument], + [{ field: "priority", operator: "lte", value: 3 }, byDocument], + [{ field: "status", operator: "ne", value: "draft" }, byDocument], + [{ field: "topics", operator: "nin", value: ["t1"] }, byDocument], + [{ field: "status", operator: "exists", value: true }, byDocument], + [{ field: "status", operator: "type", value: "string" }, byDocument], + [{ field: "topics", operator: "prefix", value: "t" }, byDocument], + [{ field: "topics", operator: "contains", value: "1" }, byDocument], + [{ field: "topics", operator: "suffix", value: "1", caseInsensitive: true }, byDocument], + ]; + + for (const statistics of [false, true]) { + if (statistics) db.exec("ANALYZE"); + for (const [filter, seek] of expectations) { + const compiled = compileMetadataFilter(parseMetadataFilter(filter), "d"); + const plan = (db.prepare(`EXPLAIN QUERY PLAN SELECT COUNT(*) FROM documents d WHERE ${compiled.sql}`) + .all(...compiled.params) as { detail: string }[]) + .map(row => row.detail) + .filter(detail => detail.includes(" mv ")); + const label = `${filter.operator}${"caseInsensitive" in filter ? " folded" : ""}${statistics ? " with statistics" : ""}`; + expect(plan.some(detail => detail.includes(seek)), `${label}: ${plan.join(" | ")}`).toBe(true); + expect(plan.some(detail => detail.includes("key=?") && !detail.includes("document_id=?")), `${label} walks a key range: ${plan.join(" | ")}`).toBe(false); + } + } + }); + test("compiled SQL never interpolates user keys or values", () => { const filter = parseMetadataFilter({ operator: "and", operands: [ { field: "key'; --", operator: "eq", value: "value'; --" }, { field: "topics", operator: "in", value: ["a'; --"] }, + { field: "topics", operator: "contains", value: "c'; --" }, + { field: "topics", operator: "suffix", value: "S'; --", caseInsensitive: true }, ], }); const compiled = compileMetadataFilter(filter, "d"); @@ -296,5 +466,7 @@ describe("compileMetadataFilter semantics", () => { expect(compiled.params).toContain("key'; --"); expect(compiled.params).toContain("value'; --"); expect(compiled.params).toContain("a'; --"); + expect(compiled.params).toContain("c'; --"); + expect(compiled.params).toContain("s'; --"); }); }); diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index fbcc0fb7a..3dbfda44c 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -22,7 +22,7 @@ import { } from "../src/store.js"; import { replaceDocumentMetadata, syncDocumentMetadata } from "../src/metadata-store.js"; import { METADATA_EXTRACTION_VERSION, type DocumentMetadata } from "../src/metadata.js"; -import type { MetadataFilter } from "../src/metadata-filter.js"; +import { parseMetadataFilter, type MetadataFilter } from "../src/metadata-filter.js"; let testDir: string; let store: Store; @@ -188,6 +188,37 @@ describe("searchVec with metadata filter", () => { expect(filtered[0]!.metadata).toEqual({ status: "published" }); }); + test("a near-ceiling filter fits both vector lookup paths and still excludes shared nonmatching copies", async () => { + store.ensureVecTable(3); + const body = "# Book\n\nDeterministic vector fixture"; + const { hash } = await insertDoc("notes", "published.md", body, { eligible: true }); + await insertDoc("notes", "draft-copy.md", body, { eligible: false }); + const timestamp = new Date().toISOString(); + + store.db.transaction(() => { + for (let sequence = 0; sequence < 20_000; sequence++) { + insertEmbedding(store.db, hash, sequence, 0, new Float32Array([1, sequence / 20_001, 0]), model, timestamp, 20_001); + } + })(); + + // 256 nodes, 64 distinct members per wide leaf: a valid filter with + // 31,490 bindings. Candidate IDs must not exhaust the remaining budget. + const members = Array.from({ length: 64 }, (_, index) => `value-${index}`); + const leaves = Array.from({ length: 246 }, () => ({ field: "absent", operator: "all", value: members })); + const groups = Array.from({ length: 8 }, (_, index) => ({ operator: "or", operands: leaves.slice(index * 32, (index + 1) * 32) })); + const metadataFilter = parseMetadataFilter({ operator: "or", operands: [{ field: "eligible", operator: "eq", value: true }, ...groups] }); + + // At 20,000 eligible chunks the exact path returns up to limit * 3 IDs. + const exactResults = await searchVec(store.db, "q", model, 500, "notes", undefined, queryEmbedding, undefined, metadataFilter); + expect(exactResults.map(result => result.displayPath)).toEqual(["notes/published.md"]); + + // The extra chunk crosses into capped global lookup. Its 4,096 candidate + // IDs used to push the final document lookup past Node's variable limit. + insertEmbedding(store.db, hash, 20_000, 0, new Float32Array([1, 1, 0]), model, timestamp, 20_001); + const fallbackResults = await searchVec(store.db, "q", model, 137, "notes", undefined, queryEmbedding, undefined, metadataFilter); + expect(fallbackResults.map(result => result.displayPath)).toEqual(["notes/published.md"]); + }); + test("returns empty when no documents are eligible", async () => { store.ensureVecTable(3); await insertEmbeddedDoc("notes", "doc.md", "# Doc", [1, 0, 0], { status: "draft" }); From 866aec2f47d41f3d318cab6aa3f410f05833731b Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Wed, 16 Sep 2026 01:27:05 -0700 Subject: [PATCH 17/82] feat(metadata): add metadata discovery store query Report keys, typed values, document coverage, collection provenance, and numeric ranges over the same extraction gate as filtered search. `filter` selects documents and `match` selects entries. Both windows run in SQL with exact totals and remainders, including past-end pages. Discovery compiles filters to uncorrelated document sets. Search retains candidate-local seeks. Entry prefix/suffix comparisons return false, not NULL, for empty strings so negation partitions the accepted value domain. Read the report in one deferred transaction. Materialize key ranking once in the type-statistics statement and reuse its selected names thereafter. Read contributing collections with those statistics. Group value counts once for both the distinct total and the value window, retaining one total carrier per type even past the end. Preserve BINARY key and value ordering. Compute even medians without overflow or loss of adjacent subnormals. Validate safe-integer options and cast the production window operands to INTEGER before adding. Check the widest planned statement against the 30,000-parameter application budget before preparing discovery queries. Provide a lean key overview and a batched per-collection overview. Ordinal zero counts each document/key once regardless of array length, without reading values or maintaining another index or cache. Tests cover aggregation, gates, filters, matches, ordering, both windows, independent boolean expectations, live-writer snapshots, lifecycle changes, actual endpoint SQL, binding limits, and structural query-work guards with and without statistics. The independent 14,276-case predicate oracle and 120 aggregate comparisons pass on both drivers after these changes. Sort contributing collection names by UTF-8 bytes after reading their aggregate. This preserves SQLite BINARY order without requiring SQLite 3.44's aggregate ORDER BY syntax. The real discovery query and the filter, discovery, and vector suites also pass on SQLite 3.43.2. Assisted-by: Claude Fable 5.1 via Pi --- src/metadata-filter.ts | 320 ++++++++--- src/metadata-store.ts | 696 +++++++++++++++++++++++- test/metadata-discovery.test.ts | 928 ++++++++++++++++++++++++++++++++ test/metadata-filter.test-d.ts | 74 +++ test/metadata-filter.test.ts | 98 +++- vitest.config.ts | 5 + 6 files changed, 2053 insertions(+), 68 deletions(-) create mode 100644 test/metadata-discovery.test.ts create mode 100644 test/metadata-filter.test-d.ts diff --git a/src/metadata-filter.ts b/src/metadata-filter.ts index 8c22a44d6..c5da72896 100644 --- a/src/metadata-filter.ts +++ b/src/metadata-filter.ts @@ -11,14 +11,24 @@ * { "field": "topics", "operator": "prefix", "value": "sql", "caseInsensitive": true } * * A condition tests one field of the record under evaluation, named by `field`, - * against `value` using `operator`. For a document the fields are its metadata - * keys. Comparison, membership, and text operators are typed by their operand: - * a string operand only ever compares against string values, a number against - * numbers, a boolean against booleans, and a mismatch never matches. + * against `value` using `operator`. The grammar evaluates two kinds of record: * - * Compilation emits correlated EXISTS/NOT EXISTS subqueries over - * `document_metadata_values` with every user value bound as a parameter. - * Metadata keys and values are data, never SQL. + * - A document, whose fields are its metadata keys. This is the filter + * (`MetadataFilter`) every search surface accepts, and it compiles to one + * subquery per condition over `document_metadata_values`, correlated for + * a few candidates or a document set for the corpus (FilterScope). + * - A metadata entry (one row of that table), whose fields are `key` and + * `value`. This is the match (`MetadataMatch`) metadata discovery accepts, + * and it compiles to a predicate over one row. + * + * Both are `MetadataPredicate`, the one recursive grammar, parameterized by + * the conditions the record admits. + * + * Comparison, membership, and text operators are typed by their operand: a + * string operand only ever compares against string values, a number against + * numbers, a boolean against booleans, and a mismatch never matches. Every + * user value binds as a parameter. Metadata keys and values are data, never + * SQL. */ import type { MetadataScalar, MetadataScalarArray, MetadataValueType } from "./metadata.js"; @@ -52,31 +62,57 @@ export type MetadataFilter = MetadataPredicate; export type MetadataFilterGroup = MetadataPredicateGroup; export type MetadataFilterNegation = MetadataPredicateNegation; +/** A predicate over a metadata entry, whose fields are `key` and `value`. */ +export type MetadataMatch = MetadataPredicate; + +/** The fields of a metadata entry. */ +export type MetadataEntryField = "key" | "value"; + /** - * One condition. `caseInsensitive` folds ASCII letters on both sides and is - * accepted only where the operand is a string or an array of strings. + * The conditions every record admits, over the fields it has. + * `caseInsensitive` folds ASCII letters on both sides and is accepted only + * where the operand is a string or an array of strings. + */ +type SharedMetadataCondition = + | { field: Field; operator: "eq" | "ne"; value: MetadataScalar; caseInsensitive?: boolean } + | { field: Field; operator: "gt" | "gte" | "lt" | "lte"; value: string | number; caseInsensitive?: boolean } + | { field: Field; operator: "in" | "nin"; value: MetadataScalarArray; caseInsensitive?: boolean } + | { field: Field; operator: "contains" | "prefix" | "suffix"; value: string; caseInsensitive?: boolean } + | { field: Field; operator: "type"; value: MetadataValueType }; + +/** + * One condition over a document. `all` and `exists` speak about the set of + * values a document holds under a key, so only a document admits them. */ export type MetadataCondition = - | { field: string; operator: "eq" | "ne"; value: MetadataScalar; caseInsensitive?: boolean } - | { field: string; operator: "gt" | "gte" | "lt" | "lte"; value: string | number; caseInsensitive?: boolean } - | { field: string; operator: "in" | "nin" | "all"; value: MetadataScalarArray; caseInsensitive?: boolean } - | { field: string; operator: "contains" | "prefix" | "suffix"; value: string; caseInsensitive?: boolean } - | { field: string; operator: "type"; value: MetadataValueType } + | SharedMetadataCondition + | { field: string; operator: "all"; value: MetadataScalarArray; caseInsensitive?: boolean } | { field: string; operator: "exists"; value: boolean }; +/** One condition over a metadata entry: its `key` (a string) or its `value` (typed). */ +export type MetadataEntryCondition = SharedMetadataCondition; + export interface CompiledMetadataFilter { sql: string; params: (string | number)[]; } -/** Raised by parseMetadataFilter with the JSON path of the failing node. */ +/** Which record a filter is evaluated against. */ +export type MetadataRecordType = "document" | "entry"; + +const ENTRY_FIELDS: ReadonlySet = new Set(["key", "value"]); + +/** Raised by parseMetadataFilter and parseMetadataMatch with the JSON path of the failing node. */ export class MetadataFilterError extends Error { readonly path: string; + /** The message without the `Invalid metadata ... at` prefix. */ + readonly detail: string; - constructor(path: string, message: string) { - super(`Invalid metadata filter at ${path}: ${message}`); + constructor(path: string, detail: string, recordType: MetadataRecordType = "document") { + super(`Invalid metadata ${recordType === "entry" ? "match" : "filter"} at ${path}: ${detail}`); this.name = "MetadataFilterError"; this.path = path; + this.detail = detail; } } @@ -107,7 +143,7 @@ const VALUE_TYPES: readonly MetadataValueType[] = ["string", "number", "boolean" // Validation // ============================================================================= -type FilterParseState = { nodes: number }; +type FilterParseState = { nodes: number; recordType: MetadataRecordType }; /** Conditions whose operand can be a string, and so can fold case. */ type CaseFoldableCondition = Exclude; @@ -119,10 +155,28 @@ type CaseFoldableCondition = Exclude METADATA_FILTER_LIMITS.maxDepth) { throw new MetadataFilterError(path, `exceeds maximum nesting depth of ${METADATA_FILTER_LIMITS.maxDepth}`); @@ -150,7 +204,7 @@ function parseFilterNode(input: unknown, path: string, depth: number, state: Fil return parseFilterNegation(node, path, depth, state); } if (CONDITION_OPERATORS.has(operator)) { - return parseFilterCondition(node, operator, path); + return parseFilterCondition(node, operator, path, state.recordType); } throw new MetadataFilterError(path, `unknown operator '${operator}' — expected one of: ${ALL_OPERATORS.join(", ")}`); @@ -201,7 +255,12 @@ function parseFilterNegation( }; } -function parseFilterCondition(node: Record, operator: string, path: string): MetadataCondition { +function parseFilterCondition( + node: Record, + operator: string, + path: string, + recordType: MetadataRecordType, +): MetadataCondition { rejectUnknownProperties(node, ["field", "operator", "value", "caseInsensitive"], path); const field = node["field"]; @@ -212,6 +271,15 @@ function parseFilterCondition(node: Record, operator: string, p throw new MetadataFilterError(path, `'field' exceeds ${METADATA_FILTER_LIMITS.maxKeyBytes} bytes`); } + if (recordType === "entry") { + if (!ENTRY_FIELDS.has(field)) { + throw new MetadataFilterError(path, `'${field}' is not a field of a metadata entry, expected 'key' or 'value'`); + } + if (operator === "exists" || operator === "all") { + throw new MetadataFilterError(path, `'${operator}' has no meaning for a single metadata entry`); + } + } + if (!("value" in node)) { throw new MetadataFilterError(path, `'${operator}' requires a 'value'`); } @@ -336,37 +404,87 @@ function rejectUnknownProperties(node: Record, allowed: string[ // ============================================================================= /** - * Compile a validated filter into one parameterized SQL predicate correlated - * against a documents-table alias (e.g. `d`). All keys and values are bound - * parameters. The caller is responsible for restricting the surrounding query - * to active documents with current, error-free metadata extraction. + * The documents a compiled filter is applied to, which decides the SQL each + * condition takes. + * + * - `candidates`: a few documents, as when search checks its results. Each + * condition is a correlated `EXISTS` that seeks the document's own rows. + * - `corpus`: every document, as when discovery counts them. Each condition + * is an uncorrelated `IN (SELECT document_id ...)` that SQLite evaluates + * once per statement and probes per document, where a correlated predicate + * would be re-run for every metadata row the statement joins. + */ +export type FilterScope = "candidates" | "corpus"; + +/** + * Compile a validated filter into one parameterized SQL predicate over a + * documents-table alias (e.g. `d`). All keys and values are bound parameters. + * The caller is responsible for restricting the surrounding query to active + * documents with current, error-free metadata extraction. + */ +export function compileMetadataFilter(filter: MetadataFilter, documentsAlias: string, scope: FilterScope = "candidates"): CompiledMetadataFilter { + const params: (string | number)[] = []; + const sql = compileFilterNode(filter, { alias: documentsAlias, scope }, params); + return { sql, params }; +} + +/** The documents-table alias a filter tests and the scope it is applied to. */ +interface FilterTarget { + alias: string; + scope: FilterScope; +} + +/** + * Compile a validated match into one parameterized SQL predicate over a + * `document_metadata_values` row alias (e.g. `mv`). A condition's `field` names + * the field of the entry its operand is tested against: `key`, which is always + * a string, or `value`, which is typed by `value_type`. The caller is + * responsible for scoping the rows to eligible documents. */ -export function compileMetadataFilter(filter: MetadataFilter, documentsAlias: string): CompiledMetadataFilter { +export function compileMetadataMatch(match: MetadataMatch, valuesAlias: string): CompiledMetadataFilter { const params: (string | number)[] = []; - const sql = compileFilterNode(filter, documentsAlias, params); + const sql = compileMatchNode(match, valuesAlias, params); return { sql, params }; } -function compileFilterNode(filter: MetadataFilter, alias: string, params: (string | number)[]): string { +/** The typed columns a condition's operand is tested against. */ +interface ValueColumns { + type: string; + text: string; + number: string; + boolean: string; +} + +/** The values of a document, as seen from inside a correlated subquery. */ +const DOCUMENT_VALUE_COLUMNS: ValueColumns = { + type: "mv.value_type", + text: "mv.text_value", + number: "mv.number_value", + boolean: "mv.boolean_value", +}; + +function compileFilterNode(filter: MetadataFilter, target: FilterTarget, params: (string | number)[]): string { + const columns = DOCUMENT_VALUE_COLUMNS; + switch (filter.operator) { case "and": case "or": { const joiner = filter.operator === "and" ? " AND " : " OR "; - return `(${filter.operands.map(operand => compileFilterNode(operand, alias, params)).join(joiner)})`; + return `(${filter.operands.map(operand => compileFilterNode(operand, target, params)).join(joiner)})`; } case "not": - return `NOT ${compileFilterNode(filter.operand, alias, params)}`; + return `NOT ${compileFilterNode(filter.operand, target, params)}`; case "exists": params.push(filter.field); return filter.value - ? buildValueExistsSql(alias, "byDocument") - : `NOT ${buildValueExistsSql(alias, "byDocument")}`; + ? buildConditionSql(target, "byDocument") + : `NOT ${buildConditionSql(target, "byDocument")}`; case "type": params.push(filter.field, filter.value); - return buildValueExistsSql(alias, "byDocument", "mv.value_type = ?"); + return buildConditionSql(target, "byDocument", `${columns.type} = ?`); case "eq": case "gt": @@ -374,12 +492,12 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin case "lt": case "lte": { const sqlOperator = { eq: "=", gt: ">", gte: ">=", lt: "<", lte: "<=" }[filter.operator]; - const columnSql = buildValueColumnSql(filter.value, filter.caseInsensitive); + const columnSql = buildOperandColumnSql(columns, filter.value, filter.caseInsensitive); params.push(filter.field, bindOperand(filter.value, filter.caseInsensitive)); - return buildValueExistsSql( - alias, + return buildConditionSql( + target, filter.operator === "eq" ? equalitySeekOf(filter.caseInsensitive) : "byDocument", - `mv.value_type = '${valueTypeOf(filter.value)}' AND ${columnSql} ${sqlOperator} ?`, + `${columns.type} = '${valueTypeOf(filter.value)}' AND ${columnSql} ${sqlOperator} ?`, ); } @@ -387,39 +505,39 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin // Key must have at least one same-type value, and no same-type value // may equal the operand. Missing keys and type mismatches do not match. const valueType = valueTypeOf(filter.value); - const columnSql = buildValueColumnSql(filter.value, filter.caseInsensitive); + const columnSql = buildOperandColumnSql(columns, filter.value, filter.caseInsensitive); params.push(filter.field); - const presentSql = buildValueExistsSql(alias, "byDocument", `mv.value_type = '${valueType}'`); + const presentSql = buildConditionSql(target, "byDocument", `${columns.type} = '${valueType}'`); params.push(filter.field, bindOperand(filter.value, filter.caseInsensitive)); - const equalSql = buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} = ?`); + const equalSql = buildConditionSql(target, equalitySeekOf(filter.caseInsensitive), `${columns.type} = '${valueType}' AND ${columnSql} = ?`); return `(${presentSql} AND NOT ${equalSql})`; } case "in": case "nin": { const valueType = valueTypeOf(filter.value[0]!); - const columnSql = buildValueColumnSql(filter.value[0]!, filter.caseInsensitive); + const columnSql = buildOperandColumnSql(columns, filter.value[0]!, filter.caseInsensitive); const placeholders = filter.value.map(() => "?").join(", "); const operands = filter.value.map(element => bindOperand(element, filter.caseInsensitive)); if (filter.operator === "in") { params.push(filter.field, ...operands); - return buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} IN (${placeholders})`); + return buildConditionSql(target, equalitySeekOf(filter.caseInsensitive), `${columns.type} = '${valueType}' AND ${columnSql} IN (${placeholders})`); } params.push(filter.field); - const presentSql = buildValueExistsSql(alias, "byDocument", `mv.value_type = '${valueType}'`); + const presentSql = buildConditionSql(target, "byDocument", `${columns.type} = '${valueType}'`); params.push(filter.field, ...operands); - const memberSql = buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} IN (${placeholders})`); + const memberSql = buildConditionSql(target, equalitySeekOf(filter.caseInsensitive), `${columns.type} = '${valueType}' AND ${columnSql} IN (${placeholders})`); return `(${presentSql} AND NOT ${memberSql})`; } case "all": { const valueType = valueTypeOf(filter.value[0]!); - const columnSql = buildValueColumnSql(filter.value[0]!, filter.caseInsensitive); + const columnSql = buildOperandColumnSql(columns, filter.value[0]!, filter.caseInsensitive); const memberSqls = filter.value.map(element => { params.push(filter.field, bindOperand(element, filter.caseInsensitive)); - return buildValueExistsSql(alias, equalitySeekOf(filter.caseInsensitive), `mv.value_type = '${valueType}' AND ${columnSql} = ?`); + return buildConditionSql(target, equalitySeekOf(filter.caseInsensitive), `${columns.type} = '${valueType}' AND ${columnSql} = ?`); }); return `(${memberSqls.join(" AND ")})`; } @@ -428,19 +546,21 @@ function compileFilterNode(filter: MetadataFilter, alias: string, params: (strin case "prefix": case "suffix": { params.push(filter.field); - const textSql = compileTextTestSql(filter.operator, filter.value, filter.caseInsensitive, params); - return buildValueExistsSql(alias, "byDocument", `mv.value_type = 'string' AND ${textSql}`); + const columnSql = buildOperandColumnSql(columns, filter.value, filter.caseInsensitive); + const textSql = compileTextTestSql(filter.operator, filter.value, columnSql, filter.caseInsensitive, params); + return buildConditionSql(target, "byDocument", `${columns.type} = 'string' AND ${textSql}`); } } } /** - * How a correlated subquery reaches one document's rows for a key. The + * How a `candidates` subquery reaches one document's rows for a key. The * covering indexes are (key, , document_id), so a condition that pins * the value column with an equality is a single probe by value. Any other * condition would walk the key's whole value range once per document (the * stats-free planner assumes a range is narrow), so it seeks the document's - * own rows through the primary key instead. + * own rows through the primary key instead. A `corpus` subquery runs once, + * so it always takes the covering index. */ type RowSeek = "byValue" | "byDocument"; @@ -450,20 +570,88 @@ function equalitySeekOf(caseInsensitive: boolean | undefined): RowSeek { } /** - * `EXISTS` over the document's rows for the key, whose parameter the caller - * binds before `conditionSql`'s. The unary `+` on the key term hides it from - * the planner, which leaves `document_id` as the only indexable term and - * makes the seek independent of `sqlite_stat1`. + * One condition over the document's rows for the key, whose parameter the + * caller binds before `conditionSql`'s. In the `candidates` scope the unary + * `+` on the key term hides it from the planner, which leaves `document_id` + * as the only indexable term and makes the seek independent of `sqlite_stat1`. */ -function buildValueExistsSql(alias: string, seek: RowSeek, conditionSql?: string): string { +function buildConditionSql(target: FilterTarget, seek: RowSeek, conditionSql?: string): string { + if (target.scope === "corpus") { + const whereSql = conditionSql ? `mv.key = ? AND ${conditionSql}` : "mv.key = ?"; + return `${target.alias}.id IN (SELECT mv.document_id FROM document_metadata_values mv WHERE ${whereSql})`; + } + const keyTermSql = seek === "byValue" ? "mv.key = ?" : "+mv.key = ?"; const whereSql = conditionSql ? `${keyTermSql} AND ${conditionSql}` : keyTermSql; - return `EXISTS (SELECT 1 FROM document_metadata_values mv WHERE mv.document_id = ${alias}.id AND ${whereSql})`; + return `EXISTS (SELECT 1 FROM document_metadata_values mv WHERE mv.document_id = ${target.alias}.id AND ${whereSql})`; +} + +function compileMatchNode(match: MetadataMatch, alias: string, params: (string | number)[]): string { + switch (match.operator) { + case "and": + case "or": { + const joiner = match.operator === "and" ? " AND " : " OR "; + return `(${match.operands.map(operand => compileMatchNode(operand, alias, params)).join(joiner)})`; + } + + case "not": + return `NOT ${compileMatchNode(match.operand, alias, params)}`; + + default: + return compileMatchCondition(match, alias, params); + } +} + +/** + * One condition over one entry. The entry's `key` field is a string with no + * other typed columns, so a number or boolean operand fails its type guard and + * matches nothing, the same outcome a type mismatch has in a filter. + */ +function compileMatchCondition(condition: MetadataEntryCondition, alias: string, params: (string | number)[]): string { + const columns: ValueColumns = condition.field === "key" + ? { type: "'string'", text: `${alias}.key`, number: "NULL", boolean: "NULL" } + : { type: `${alias}.value_type`, text: `${alias}.text_value`, number: `${alias}.number_value`, boolean: `${alias}.boolean_value` }; + + switch (condition.operator) { + case "type": + params.push(condition.value); + return `${columns.type} = ?`; + + case "eq": + case "ne": + case "gt": + case "gte": + case "lt": + case "lte": { + const sqlOperator = { eq: "=", ne: "<>", gt: ">", gte: ">=", lt: "<", lte: "<=" }[condition.operator]; + const columnSql = buildOperandColumnSql(columns, condition.value, condition.caseInsensitive); + params.push(bindOperand(condition.value, condition.caseInsensitive)); + return `(${columns.type} = '${valueTypeOf(condition.value)}' AND ${columnSql} ${sqlOperator} ?)`; + } + + case "in": + case "nin": { + const valueType = valueTypeOf(condition.value[0]!); + const columnSql = buildOperandColumnSql(columns, condition.value[0]!, condition.caseInsensitive); + const placeholders = condition.value.map(() => "?").join(", "); + const membership = condition.operator === "in" ? "IN" : "NOT IN"; + params.push(...condition.value.map(element => bindOperand(element, condition.caseInsensitive))); + return `(${columns.type} = '${valueType}' AND ${columnSql} ${membership} (${placeholders}))`; + } + + case "contains": + case "prefix": + case "suffix": { + const columnSql = buildOperandColumnSql(columns, condition.value, condition.caseInsensitive); + const textSql = compileTextTestSql(condition.operator, condition.value, columnSql, condition.caseInsensitive, params); + return `(${columns.type} = 'string' AND ${textSql})`; + } + } } /** - * Substring tests over the string column. Prefix and suffix compare UTF-8 - * bytes, with the operand's byte length bound from JavaScript: SQLite's text + * Substring tests over a string column. Prefix and suffix compare UTF-8 bytes, + * with the operand's byte length bound from JavaScript: SQLite's text * `length()` and `substr()` stop at an embedded NUL, and blobs do not. UTF-8 * is self-synchronizing, so a byte-prefix (or byte-suffix) of a whole operand * is exactly a character-prefix (or -suffix). @@ -471,10 +659,10 @@ function buildValueExistsSql(alias: string, seek: RowSeek, conditionSql?: string function compileTextTestSql( operator: "contains" | "prefix" | "suffix", text: string, + columnSql: string, caseInsensitive: boolean | undefined, params: (string | number)[], ): string { - const columnSql = buildValueColumnSql(text, caseInsensitive); const operand = caseInsensitive ? foldAsciiCase(text) : text; if (operator === "contains") { @@ -483,9 +671,11 @@ function compileTextTestSql( } params.push(utf8ByteLengthOf(operand), operand); + // SQLite returns NULL for a substring of an empty BLOB. Each primitive + // must return a boolean so entry matches and their negations partition rows. return operator === "prefix" - ? `substr(CAST(${columnSql} AS BLOB), 1, ?) = CAST(? AS BLOB)` - : `substr(CAST(${columnSql} AS BLOB), -?) = CAST(? AS BLOB)`; + ? `COALESCE(substr(CAST(${columnSql} AS BLOB), 1, ?) = CAST(? AS BLOB), 0)` + : `COALESCE(substr(CAST(${columnSql} AS BLOB), -?) = CAST(? AS BLOB), 0)`; } const utf8Encoder = new TextEncoder(); @@ -499,10 +689,10 @@ function valueTypeOf(scalar: MetadataScalar): MetadataValueType { } /** The typed column an operand compares against, folded when the condition ignores case. */ -function buildValueColumnSql(scalar: MetadataScalar, caseInsensitive: boolean | undefined): string { - if (typeof scalar === "number") return "mv.number_value"; - if (typeof scalar === "boolean") return "mv.boolean_value"; - return caseInsensitive ? "lower(mv.text_value)" : "mv.text_value"; +function buildOperandColumnSql(columns: ValueColumns, scalar: MetadataScalar, caseInsensitive: boolean | undefined): string { + if (typeof scalar === "number") return columns.number; + if (typeof scalar === "boolean") return columns.boolean; + return caseInsensitive ? `lower(${columns.text})` : columns.text; } /** Booleans bind as 0/1. Case-insensitive strings bind folded the same way SQLite's `lower()` folds the column. */ diff --git a/src/metadata-store.ts b/src/metadata-store.ts index 704b2268b..0bf515498 100644 --- a/src/metadata-store.ts +++ b/src/metadata-store.ts @@ -14,13 +14,25 @@ * for filtering. */ -import type { Database } from "./db.js"; +import * as buffer from "node:buffer"; + +import type { Database, SQLiteValue } from "./db.js"; import { extractDocumentMetadata, METADATA_EXTRACTION_VERSION, type DocumentMetadata, type MetadataExtractionResult, + type MetadataScalar, + type MetadataValueType, } from "./metadata.js"; +import { + compileMetadataFilter, + compileMetadataMatch, + parseMetadataMatch, + type CompiledMetadataFilter, + type MetadataFilter, + type MetadataMatch, +} from "./metadata-filter.js"; // ============================================================================= // Schema @@ -210,3 +222,685 @@ export function parseMetadataJson(metadataJson: string | null | undefined): Docu return {}; } } + +// ============================================================================= +// Discovery +// ============================================================================= + +export interface ListMetadataOptions { + /** Restrict to these collections. Undefined means every collection in the index. */ + collection?: string | string[]; + /** + * Report only the metadata entries matching this condition. Same grammar as + * `filter`, evaluated against each entry: a condition's `field` names the + * entry's `key` or `value`. Undefined reports every entry. + */ + match?: MetadataMatch; + /** Count only documents matching this filter. Same AST as search. */ + filter?: MetadataFilter; + /** Keys reported (default 50). `Infinity` removes the window. */ + keyLimit?: number; + /** Keys skipped before the window, in report order (default 0). */ + keyOffset?: number; + /** Values reported per key and type (default 10). `Infinity` removes the window. */ + valueLimit?: number; + /** Values skipped per key and type before the window, in `sort` order (default 0). */ + valueOffset?: number; + /** Order of values within a key (default "count"). */ + sort?: "count" | "value"; + /** Drop values held by fewer documents than this (default 1). */ + minCount?: number; +} + +export interface ListMetadataResult { + /** Active documents in scope. */ + documents: number; + /** + * Documents in scope that pass `filter`. Present only when a filter was + * given, and then the denominator for every coverage count. + */ + filteredDocuments?: number; + /** Keys with a matching entry, before the key window. */ + totalKeys: number; + /** The key window, by documents descending then key ascending. */ + keys: MetadataKeySummary[]; + /** Keys after the window: `totalKeys - keyOffset - keys.length`, floored at 0. */ + remainingKeys: number; +} + +export interface MetadataKeySummary { + key: string; + /** Distinct documents holding a matching entry under any type. */ + documents: number; + /** One entry per value_type present. Length > 1 is a type conflict. */ + types: MetadataKeyTypeSummary[]; +} + +export interface MetadataKeyTypeSummary { + type: MetadataValueType; + /** + * True when any document holds more than one value for this key. A + * one-element array is indistinguishable from a scalar in the index. + */ + multiValued: boolean; + /** Distinct documents holding a matching value of this type. */ + documents: number; + /** Distinct matching values that meet `minCount`. */ + distinctValues: number; + /** The value window, narrowed by `match` and `minCount`, ordered by `sort`. */ + values: MetadataValueCount[]; + /** Distinct values after the window: `distinctValues - valueOffset - values.length`, floored at 0. */ + remainingValues: number; + /** Numbers only. Computed over every matching value row, so array elements each count. */ + range?: { min: number; median: number; max: number }; + /** Collections contributing a matching value of this type, ascending. */ + collections: string[]; +} + +export interface MetadataValueCount { + value: MetadataScalar; + documents: number; +} + +/** The light form of a key summary for status views: name, coverage, and types, no values. */ +export interface MetadataKeyOverview { + key: string; + /** Distinct documents declaring the key with any type. */ + documents: number; + /** By documents descending then name. Length > 1 is a type conflict. */ + types: MetadataValueType[]; +} + +/** Exact vocabulary size and a bounded key overview, computed together. */ +export interface MetadataOverview { + totalKeys: number; + keys: MetadataKeyOverview[]; +} + +export const DEFAULT_METADATA_KEY_LIMIT = 50; +export const DEFAULT_METADATA_VALUE_LIMIT = 10; + +/** + * Application budget for the bound parameters of one discovery statement, under + * SQLite's default variable limit (32,766) with headroom. A filter and a match + * each pass the parser's own limits independently, and the match is bound + * twice (once to rank keys, once to aggregate), so the combination is checked + * here, before any statement is prepared. + */ +export const METADATA_SQL_BINDING_BUDGET = 30_000; + +/** Raised by listMetadata when an option is outside its domain. */ +export class MetadataOptionError extends Error { + readonly option: string; + + constructor(option: string, expected: string, received: unknown) { + super(`Invalid ${option}: expected ${expected}, received ${formatReceived(received)}`); + this.name = "MetadataOptionError"; + this.option = option; + } +} + +/** Raised by listMetadata when `filter` and `match` together bind more SQL parameters than the budget. */ +export class MetadataBindingBudgetError extends Error { + readonly bindings: number; + readonly budget: number; + + constructor(bindings: number, budget: number) { + super(`filter and match together bind ${bindings} SQL parameters, over the budget of ${budget}. Shorten their membership lists.`); + this.name = "MetadataBindingBudgetError"; + this.bindings = bindings; + this.budget = budget; + } +} + +/** The window and ordering options with defaults applied and domains checked. */ +interface ResolvedWindow { + keyLimit: number; + keyOffset: number; + valueLimit: number; + valueOffset: number; + sort: "count" | "value"; + minCount: number; +} + +/** + * The value rows discovery aggregates over. `withSql` defines the `eligible` + * CTE (and, once a key window is applied, `selected_keys`); `fromSql` joins + * `document_metadata_values mv` to them, narrowed by the match when one is + * set. Each query supplies its own SELECT list, WHERE, and GROUP BY around + * these two parts. + */ +interface Region { + withSql: string; + withParams: SQLiteValue[]; + fromSql: string; + fromParams: SQLiteValue[]; +} + +type ValueRow = { + key: string; + value_type: MetadataValueType; + text_value: string | null; + number_value: number | null; + boolean_value: number | null; +}; + +/** + * Summarize metadata keys, types, and value counts for the documents in + * scope. Discovery sees exactly what filtering sees: the same extraction gate, + * active-document rule, and collection scope, so every value reported here is + * a value an `eq` filter can match. + * + * `filter` selects which documents are counted. `match` selects which of their + * metadata entries are reported, with the same grammar evaluated per entry. + * Counts are documents, not values: a document with `topics: [a, b]` + * contributes one to each. + * + * The key window limits which keys receive detailed aggregation and the value + * window limits the values returned per key and type. Exact counts, medians, + * and ordering still process the relevant rows. These windows do not bound + * database work, temporary storage, or the contributing collection lists. + * + * The report is read inside one deferred transaction, so every count comes + * from the same database snapshot even while another connection writes. + */ +export function listMetadata(db: Database, options: ListMetadataOptions = {}): ListMetadataResult { + const window = resolveWindow(options); + // A match built in code gets the same checks as one parsed from JSON. + const match = options.match ? compileMetadataMatch(parseMetadataMatch(options.match), "mv") : undefined; + return db.transaction(() => readMetadataReport(db, options, match, window))(); +} + +function readMetadataReport( + db: Database, + options: ListMetadataOptions, + match: CompiledMetadataFilter | undefined, + window: ResolvedWindow, +): ListMetadataResult { + const collectionNames = options.collection === undefined ? undefined : [options.collection].flat(); + const eligible = buildEligibleCte(collectionNames, options.filter); + const matchedRegion = buildRegion(eligible, match); + const keyWindowRegion = buildKeyWindowRegion(matchedRegion, window.keyLimit, window.keyOffset); + assertBindingBudget(keyWindowRegion, matchedRegion); + + const result: ListMetadataResult = { + documents: countActiveDocuments(db, collectionNames), + totalKeys: 0, + keys: [], + remainingKeys: 0, + }; + if (options.filter) result.filteredDocuments = countEligibleDocuments(db, eligible); + + result.totalKeys = countKeys(db, matchedRegion); + if (result.totalKeys === 0) return result; + + const typeStats = queryTypeStats(db, keyWindowRegion); + if (typeStats.length === 0) return result; + + // Carry the selected names through the report. MATERIALIZED prevents repeat + // ranking within one statement, but cannot share it across statements. + const keyNames = [...new Set(typeStats.map(row => row.key))]; + const region = narrowRegionToKeys(matchedRegion, keyNames); + const medianByKey = queryNumberMedians(db, region); + const typeSummaryByKeyType = new Map(); + const typeSummariesByKey = new Map(); + + for (const row of typeStats) { + const typeSummary: MetadataKeyTypeSummary = { + type: row.value_type, + multiValued: false, + documents: row.documents, + distinctValues: 0, + values: [], + remainingValues: 0, + collections: (JSON.parse(row.collections) as string[]).sort(compareBinary), + }; + if (row.value_type === "number") { + typeSummary.range = { min: row.min_value!, median: medianByKey.get(row.key)!, max: row.max_value! }; + } + typeSummaryByKeyType.set(`${row.key}\0${row.value_type}`, typeSummary); + + const typeSummaries = typeSummariesByKey.get(row.key) ?? []; + typeSummaries.push(typeSummary); + typeSummariesByKey.set(row.key, typeSummaries); + } + + // Array-ness describes the key, not the matched entries, so it is read over + // every entry of the keys in the window. + const keyRegion = buildRegion(eligible, buildKeySetPredicate([...typeSummariesByKey.keys()])); + for (const row of queryTypeArrayness(db, keyRegion)) { + const typeSummary = typeSummaryByKeyType.get(`${row.key}\0${row.value_type}`); + if (typeSummary) typeSummary.multiValued = row.multi_valued === 1; + } + + for (const row of queryValueCounts(db, region, window)) { + const typeSummary = typeSummaryByKeyType.get(`${row.key}\0${row.value_type}`); + if (!typeSummary) continue; + typeSummary.distinctValues = row.distinct_values; + if (row.in_window) typeSummary.values.push({ value: scalarOf(row), documents: row.documents }); + } + + for (const typeSummary of typeSummaryByKeyType.values()) { + typeSummary.remainingValues = Math.max(0, typeSummary.distinctValues - window.valueOffset - typeSummary.values.length); + } + + // Keys arrive in window rank order and stay in it. A document holds a key + // under exactly one type (arrays are homogeneous), so per-type document + // counts partition the key's documents. + for (const [key, typeSummaries] of typeSummariesByKey) { + typeSummaries.sort(compareTypeSummaries); + result.keys.push({ + key, + documents: typeSummaries.reduce((sum, typeSummary) => sum + typeSummary.documents, 0), + types: typeSummaries, + }); + } + result.remainingKeys = Math.max(0, result.totalKeys - window.keyOffset - result.keys.length); + + return result; +} + +/** + * Key names, coverage, and types for the documents in scope, in coverage + * order, windowed to the first `keyLimit` keys. One GROUP BY, no values: what + * `collection list`, `status`, and the MCP status tool print so a first look + * reveals that metadata exists. `countMetadataKeys` gives the total. + */ +export function listMetadataKeys(db: Database, collectionNames?: string[], keyLimit: number = DEFAULT_METADATA_KEY_LIMIT): MetadataKeyOverview[] { + return getMetadataOverview(db, collectionNames, keyLimit).keys; +} + +export function getMetadataOverview(db: Database, collectionNames?: string[], keyLimit: number = DEFAULT_METADATA_KEY_LIMIT): MetadataOverview { + return queryMetadataOverviews(db, buildEligibleCte(collectionNames, undefined), "''", keyLimit).get("") ?? { totalKeys: 0, keys: [] }; +} + +/** Read every collection's overview in one pass, not two queries per collection. */ +export function listMetadataCollectionSummaries(db: Database, keyLimit: number = DEFAULT_METADATA_KEY_LIMIT): Map { + return queryMetadataOverviews(db, buildEligibleCte(undefined, undefined), "e.collection", keyLimit); +} + +interface MetadataKeyCoverage { + collection: string; + key: string; + documents: number; + string_documents: number; + number_documents: number; + boolean_documents: number; + total_keys: number; +} + +function queryMetadataOverviews(db: Database, eligible: Region, collectionSql: string, keyLimit: number): Map { + const region = buildRegion(eligible); + // Ordinal zero represents a document/key once. Extraction guarantees one + // homogeneous type per key, so arrays need neither value reads nor DISTINCT. + const coverages = db.prepare(` + ${region.withSql}, + key_coverages AS ( + SELECT ${collectionSql} AS collection, mv.key, COUNT(*) AS documents, + SUM(mv.value_type = 'string') AS string_documents, + SUM(mv.value_type = 'number') AS number_documents, + SUM(mv.value_type = 'boolean') AS boolean_documents, + COUNT(*) OVER (PARTITION BY ${collectionSql}) AS total_keys, + ROW_NUMBER() OVER (PARTITION BY ${collectionSql} ORDER BY COUNT(*) DESC, mv.key) AS rank + ${region.fromSql} + WHERE mv.ordinal = 0 + GROUP BY ${collectionSql}, mv.key + ) + SELECT * FROM key_coverages WHERE rank <= ? OR ? = -1 ORDER BY collection, rank + `).all(...region.withParams, ...region.fromParams, Number.isFinite(keyLimit) ? keyLimit : -1, Number.isFinite(keyLimit) ? keyLimit : -1) as MetadataKeyCoverage[]; + + const overviews = new Map(); + for (const coverage of coverages) { + const overview = overviews.get(coverage.collection) ?? { totalKeys: coverage.total_keys, keys: [] }; + const typeCounts: [MetadataValueType, number][] = [["string", coverage.string_documents], ["number", coverage.number_documents], ["boolean", coverage.boolean_documents]]; + typeCounts.sort((left, right) => right[1] - left[1] || compareBinary(left[0], right[0])); + overview.keys.push({ key: coverage.key, documents: coverage.documents, types: typeCounts.filter(([, count]) => count > 0).map(([type]) => type) }); + overviews.set(coverage.collection, overview); + } + return overviews; +} + +function compareTypeSummaries(a: MetadataKeyTypeSummary, b: MetadataKeyTypeSummary): number { + return b.documents - a.documents || compareBinary(a.type, b.type); +} + +/** SQLite BINARY order compares UTF-8 bytes, not JavaScript's UTF-16 code units. */ +function compareBinary(left: string, right: string): number { + return buffer.Buffer.compare(buffer.Buffer.from(left), buffer.Buffer.from(right)); +} + +/** Distinct metadata keys declared by active, extracted documents in scope. */ +export function countMetadataKeys(db: Database, collectionNames?: string[]): number { + return countKeys(db, buildRegion(buildEligibleCte(collectionNames, undefined))); +} + +/** Active, extracted documents in scope that declare at least one metadata key. */ +export function countDocumentsWithMetadata(db: Database, collectionNames?: string[]): number { + const region = buildRegion(buildEligibleCte(collectionNames, undefined)); + const row = db.prepare(` + ${region.withSql} + SELECT COUNT(DISTINCT mv.document_id) AS c + ${region.fromSql} + `).get(...region.withParams, ...region.fromParams) as { c: number }; + return row.c; +} + +// ----------------------------------------------------------------------------- +// Options +// ----------------------------------------------------------------------------- + +function resolveWindow(options: ListMetadataOptions): ResolvedWindow { + return { + keyLimit: readLimit(options.keyLimit, "keyLimit", DEFAULT_METADATA_KEY_LIMIT), + keyOffset: readOffset(options.keyOffset, "keyOffset"), + valueLimit: readLimit(options.valueLimit, "valueLimit", DEFAULT_METADATA_VALUE_LIMIT), + valueOffset: readOffset(options.valueOffset, "valueOffset"), + sort: readSort(options.sort), + minCount: readPositiveInteger(options.minCount, "minCount", 1), + }; +} + +// Safe integers are exactly representable in JavaScript. The drivers may bind +// them as REAL, so SQL arithmetic must cast explicitly when it needs INTEGER. +function readLimit(value: unknown, option: string, fallback: number): number { + if (value === undefined) return fallback; + if (value === Infinity) return Infinity; + if (isSafeInteger(value) && value >= 1) return value; + throw new MetadataOptionError(option, "a positive integer or Infinity", value); +} + +function readOffset(value: unknown, option: string): number { + if (value === undefined) return 0; + if (isSafeInteger(value) && value >= 0) return value; + throw new MetadataOptionError(option, "a non-negative integer", value); +} + +function readPositiveInteger(value: unknown, option: string, fallback: number): number { + if (value === undefined) return fallback; + if (isSafeInteger(value) && value >= 1) return value; + throw new MetadataOptionError(option, "a positive integer", value); +} + +function isSafeInteger(value: unknown): value is number { + return Number.isSafeInteger(value); +} + +function readSort(value: unknown): "count" | "value" { + if (value === undefined) return "count"; + if (value === "count" || value === "value") return value; + throw new MetadataOptionError("sort", "'count' or 'value'", value); +} + +function formatReceived(value: unknown): string { + if (typeof value === "string") return JSON.stringify(value); + if (typeof value === "number" || typeof value === "boolean" || value === null) return String(value); + return typeof value; +} + +// ----------------------------------------------------------------------------- +// Regions +// ----------------------------------------------------------------------------- + +/** + * Documents that filtered search can see: active, with a current, error-free + * extraction, in scope, and passing the filter. Lists bind as one JSON + * parameter so the SQLite variable limit never applies. The filter is + * compiled for the corpus: every statement that reads this CTE joins it to + * metadata rows, and a correlated predicate would be re-run for each of + * them, where a document set is built once per statement and probed. + */ +function buildEligibleCte(collectionNames: string[] | undefined, filter: MetadataFilter | undefined): Region { + const withParams: SQLiteValue[] = []; + let withSql = ` + WITH eligible AS ( + SELECT d.id AS document_id, d.collection + FROM documents d + JOIN document_metadata dm ON dm.document_id = d.id + WHERE d.active = 1 + AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} + AND dm.extraction_error IS NULL`; + + if (collectionNames) { + withSql += ` + AND d.collection IN (SELECT value FROM json_each(?))`; + withParams.push(JSON.stringify(collectionNames)); + } + + if (filter) { + const compiledFilter = compileMetadataFilter(filter, "d", "corpus"); + withSql += ` + AND ${compiledFilter.sql}`; + withParams.push(...compiledFilter.params); + } + + withSql += ` + )`; + + return { withSql, withParams, fromSql: "", fromParams: [] }; +} + +/** + * Join eligible documents' values, narrowed to the entries an optional + * predicate over `mv` admits. The predicate lives in the JOIN so queries can + * add their own WHERE. + */ +function buildRegion(eligible: Region, predicate?: CompiledMetadataFilter): Region { + const region: Region = { + withSql: eligible.withSql, + withParams: [...eligible.withParams], + fromSql: ` + FROM document_metadata_values mv + JOIN eligible e ON e.document_id = mv.document_id`, + fromParams: [], + }; + + if (predicate) { + region.fromSql += ` + AND ${predicate.sql}`; + region.fromParams.push(...predicate.params); + } + + return region; +} + +/** + * Narrow a region to one window of its keys, in report order (documents + * descending, then key in BINARY collation), each carrying its `rank` so the + * order survives the joins and GROUP BYs that read it. The window is a CTE + * every later aggregation joins. It is MATERIALIZED because a small LIMIT + * otherwise invites the planner to inline it as a coroutine and re-run the + * ranking once per value row. SQLite reads `LIMIT -1` as no limit. + */ +function buildKeyWindowRegion(region: Region, keyLimit: number, keyOffset: number): Region { + const withSql = `${region.withSql}, + selected_keys AS MATERIALIZED ( + SELECT mv.key, ROW_NUMBER() OVER (ORDER BY COUNT(DISTINCT mv.document_id) DESC, mv.key ASC) AS rank + ${region.fromSql} + GROUP BY mv.key + ORDER BY rank + LIMIT ? OFFSET ? + )`; + + return { + withSql, + withParams: [...region.withParams, ...region.fromParams, Number.isFinite(keyLimit) ? keyLimit : -1, keyOffset], + fromSql: `${region.fromSql} + JOIN selected_keys sk ON sk.key = mv.key`, + fromParams: [...region.fromParams], + }; +} + +/** Reuse the report's selected keys without repeating coverage and ranking. */ +function narrowRegionToKeys(region: Region, keyNames: string[]): Region { + const predicate = buildKeySetPredicate(keyNames); + return { + ...region, + fromSql: `${region.fromSql} AND ${predicate.sql}`, + fromParams: [...region.fromParams, ...predicate.params], + }; +} + +/** Every entry of these keys, bound as one JSON parameter. */ +function buildKeySetPredicate(keyNames: string[]): CompiledMetadataFilter { + return { sql: "mv.key IN (SELECT value FROM json_each(?))", params: [JSON.stringify(keyNames)] }; +} + +/** Selected names as JSON, minCount, offset, and the two endpoint operands. */ +const VALUE_QUERY_BINDINGS = 5; + +function assertBindingBudget(keyWindowRegion: Region, matchedRegion: Region): void { + const bindings = Math.max( + keyWindowRegion.withParams.length + keyWindowRegion.fromParams.length, + matchedRegion.withParams.length + matchedRegion.fromParams.length + VALUE_QUERY_BINDINGS, + ); + if (bindings > METADATA_SQL_BINDING_BUDGET) throw new MetadataBindingBudgetError(bindings, METADATA_SQL_BINDING_BUDGET); +} + +// ----------------------------------------------------------------------------- +// Queries +// ----------------------------------------------------------------------------- + +function countActiveDocuments(db: Database, collectionNames: string[] | undefined): number { + let sql = `SELECT COUNT(*) AS c FROM documents d WHERE d.active = 1`; + const params: SQLiteValue[] = []; + if (collectionNames) { + sql += ` AND d.collection IN (SELECT value FROM json_each(?))`; + params.push(JSON.stringify(collectionNames)); + } + const row = db.prepare(sql).get(...params) as { c: number }; + return row.c; +} + +function countEligibleDocuments(db: Database, eligible: Region): number { + const row = db.prepare(`${eligible.withSql} SELECT COUNT(*) AS c FROM eligible`).get(...eligible.withParams) as { c: number }; + return row.c; +} + +function countKeys(db: Database, region: Region): number { + const row = db.prepare(` + ${region.withSql} + SELECT COUNT(DISTINCT mv.key) AS c + ${region.fromSql} + `).get(...region.withParams, ...region.fromParams) as { c: number }; + return row.c; +} + +type TypeStatsRow = { key: string; value_type: MetadataValueType; documents: number; min_value: number | null; max_value: number | null; collections: string }; + +/** + * Per key and type over a key-window region, in the window's rank order. + * Collection names are sorted after reading: aggregate ORDER BY would + * require SQLite 3.44, newer than QMD's declared Bun minimum. + */ +function queryTypeStats(db: Database, region: Region): TypeStatsRow[] { + return db.prepare(` + ${region.withSql} + SELECT mv.key, mv.value_type, + COUNT(DISTINCT mv.document_id) AS documents, + MIN(mv.number_value) AS min_value, + MAX(mv.number_value) AS max_value, + json_group_array(DISTINCT e.collection) AS collections + ${region.fromSql} + GROUP BY mv.key, mv.value_type + ORDER BY sk.rank, mv.value_type + `).all(...region.withParams, ...region.fromParams) as TypeStatsRow[]; +} + +type TypeArraynessRow = { key: string; value_type: MetadataValueType; multi_valued: number }; + +function queryTypeArrayness(db: Database, keyRegion: Region): TypeArraynessRow[] { + return db.prepare(` + ${keyRegion.withSql} + SELECT mv.key, mv.value_type, MAX(mv.ordinal) > 0 AS multi_valued + ${keyRegion.fromSql} + GROUP BY mv.key, mv.value_type + `).all(...keyRegion.withParams, ...keyRegion.fromParams) as TypeArraynessRow[]; +} + +type MiddleValueRow = { key: string; number_value: number }; + +/** + * Median over value rows per key: the middle row for odd counts, the midpoint + * of the two middle rows for even. The midpoint is taken in JavaScript because + * SQL's AVG sums first and the sum of two finite doubles can overflow. + */ +function queryNumberMedians(db: Database, region: Region): Map { + const middleRows = db.prepare(` + ${region.withSql} + SELECT key, number_value + FROM ( + SELECT mv.key, mv.number_value, + ROW_NUMBER() OVER (PARTITION BY mv.key ORDER BY mv.number_value) AS position, + COUNT(*) OVER (PARTITION BY mv.key) AS total + ${region.fromSql} + WHERE mv.value_type = 'number' + ) + WHERE position IN ((total + 1) / 2, (total + 2) / 2) + ORDER BY key, number_value + `).all(...region.withParams, ...region.fromParams) as MiddleValueRow[]; + + const medianByKey = new Map(); + for (const row of middleRows) { + const lower = medianByKey.get(row.key); + medianByKey.set(row.key, lower === undefined ? row.number_value : midpointOf(lower, row.number_value)); + } + return medianByKey; +} + +/** + * The correctly rounded midpoint when the sum is representable, which also + * keeps adjacent subnormals exact. Halving each side first is reserved for the + * overflow region, where both operands are far from underflow. + */ +function midpointOf(lower: number, upper: number): number { + const sum = lower + upper; + if (Number.isFinite(sum)) return sum / 2; + return lower / 2 + upper / 2; +} + +type ValueCountRow = ValueRow & { documents: number; distinct_values: number; in_window: number }; + +/** + * Group distinct values once for both the total and the window. The first + * row of each partition carries its total even when the window is past the + * end. Only rows marked in_window become returned values. + * Only one typed column is non-null per partition, so ordering all three is stable. + */ +function queryValueCounts(db: Database, region: Region, window: ResolvedWindow): ValueCountRow[] { + const valueOrder = "mv.text_value, mv.number_value, mv.boolean_value"; + const rankOrder = window.sort === "count" ? `COUNT(DISTINCT mv.document_id) DESC, ${valueOrder}` : valueOrder; + const params = [...region.withParams, ...region.fromParams, window.minCount, window.valueOffset]; + let windowSql = "rank > ?"; + // Both drivers may bind REALs. Cast before adding so the endpoint remains + // exact even when it exceeds JavaScript's safe-integer range. + if (Number.isFinite(window.valueLimit)) { + windowSql += " AND rank <= CAST(? AS INTEGER) + CAST(? AS INTEGER)"; + params.push(window.valueOffset, window.valueLimit); + } + + return db.prepare(` + ${region.withSql}, + value_counts AS ( + SELECT mv.key, mv.value_type, mv.text_value, mv.number_value, mv.boolean_value, + COUNT(DISTINCT mv.document_id) AS documents, + COUNT(*) OVER (PARTITION BY mv.key, mv.value_type) AS distinct_values, + ROW_NUMBER() OVER (PARTITION BY mv.key, mv.value_type ORDER BY ${rankOrder}) AS rank + ${region.fromSql} + GROUP BY mv.key, mv.value_type, mv.text_value, mv.number_value, mv.boolean_value + HAVING COUNT(DISTINCT mv.document_id) >= ? + ), + value_window AS ( + SELECT *, (${windowSql}) AS in_window FROM value_counts + ) + SELECT key, value_type, text_value, number_value, boolean_value, documents, distinct_values, in_window + FROM value_window + WHERE rank = 1 OR in_window + ORDER BY key, value_type, rank + `).all(...params) as ValueCountRow[]; +} + +function scalarOf(row: ValueRow): MetadataScalar { + if (row.value_type === "string") return row.text_value!; + if (row.value_type === "number") return row.number_value!; + return row.boolean_value === 1; +} diff --git a/test/metadata-discovery.test.ts b/test/metadata-discovery.test.ts new file mode 100644 index 000000000..954886c67 --- /dev/null +++ b/test/metadata-discovery.test.ts @@ -0,0 +1,928 @@ +/** + * metadata-discovery.test.ts - Store-level metadata discovery: key summaries, + * per-type value windows, the match over metadata entries, and scope and gate + * agreement with filtered search. + */ + +import { describe, test, expect, beforeAll, afterAll, beforeEach, afterEach } from "vitest"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { + createStore, + insertContent, + insertDocument, + hashContent, + searchFTS, + type Store, +} from "../src/store.js"; +import { + listMetadata, + listMetadataKeys, + getMetadataOverview, + listMetadataCollectionSummaries, + METADATA_SQL_BINDING_BUDGET, + MetadataBindingBudgetError, + MetadataOptionError, + replaceDocumentMetadata, + type ListMetadataOptions, + type MetadataKeySummary, +} from "../src/metadata-store.js"; +import type { Database, SQLiteValue } from "../src/db.js"; +import { compileMetadataFilter, compileMetadataMatch, MetadataFilterError, type MetadataFilter, type MetadataMatch } from "../src/metadata-filter.js"; +import { METADATA_EXTRACTION_VERSION, type DocumentMetadata } from "../src/metadata.js"; + +let testDir: string; +let store: Store; + +beforeAll(async () => { + testDir = await mkdtemp(join(tmpdir(), "qmd-metadata-discovery-")); +}); + +afterAll(async () => { + await rm(testDir, { recursive: true, force: true }); +}); + +beforeEach(() => { + const dbPath = join(testDir, `test-${Date.now()}-${Math.random().toString(36).slice(2)}.sqlite`); + store = createStore(dbPath); +}); + +afterEach(() => { + store.close(); +}); + +let documentCounter = 0; + +/** Insert an active document with extracted metadata. Every body contains "doc" so FTS can reach it. */ +async function insertMetadataDoc(collection: string, metadata: DocumentMetadata): Promise { + documentCounter += 1; + const path = `doc-${documentCounter}.md`; + const content = `# doc ${documentCounter}\n\nbody of doc ${documentCounter}\n`; + const now = new Date().toISOString(); + const hash = await hashContent(content); + insertContent(store.db, hash, content, now); + const documentId = insertDocument(store.db, collection, path, path, hash, now, now); + replaceDocumentMetadata(store.db, documentId, { metadata, extractionVersion: METADATA_EXTRACTION_VERSION }); + return documentId; +} + +/** The store's database with every prepared statement's SQL reported before it runs. */ +function observeStatements(db: Database, onPrepare: (sql: string) => void, onAll?: (sql: string, params: SQLiteValue[]) => void): Database { + return new Proxy(db, { + get(target, property) { + if (property === "prepare") { + return (sql: string) => { + onPrepare(sql); + const statement = target.prepare(sql); + if (!onAll) return statement; + + return new Proxy(statement, { + get(prepared, member) { + if (member === "all") return (...params: SQLiteValue[]) => { + onAll(sql, params); + return prepared.all(...params); + }; + const value = prepared[member as keyof typeof prepared]; + return typeof value === "function" ? value.bind(prepared) : value; + }, + }); + }; + } + // Bun's Database keeps private fields, so its methods must run against the real instance. + const member = target[property as keyof Database]; + return typeof member === "function" ? member.bind(target) : member; + }, + }); +} + +/** The statement's query plan. Placeholders bind NULL, which is enough to plan. */ +function planOf(sql: string): string[] { + const placeholders = Array.from(sql.matchAll(/\?/g), () => null); + const rows = store.db.prepare(`EXPLAIN QUERY PLAN ${sql}`).all(...placeholders) as { detail: string }[]; + return rows.map(row => row.detail); +} + +function summaryOf(keys: MetadataKeySummary[], key: string): MetadataKeySummary { + const summary = keys.find(candidate => candidate.key === key); + if (!summary) throw new Error(`key ${key} missing from ${keys.map(candidate => candidate.key).join(", ")}`); + return summary; +} + +function valuesOf(keys: MetadataKeySummary[], key: string): [string | number | boolean, number][] { + return summaryOf(keys, key).types[0]!.values.map(count => [count.value, count.documents]); +} + +describe("listMetadata counting", () => { + test("counts documents, not values, and reports coverage per key", async () => { + await insertMetadataDoc("notes", { topics: ["a", "b"], status: "draft" }); + await insertMetadataDoc("notes", { topics: ["a"], status: "published" }); + await insertMetadataDoc("notes", { status: "published" }); + + const result = listMetadata(store.db); + + expect(result.documents).toBe(3); + expect(result.filteredDocuments).toBeUndefined(); + expect(result.keys.map(summary => summary.key)).toEqual(["status", "topics"]); + + const topics = summaryOf(result.keys, "topics"); + expect(topics.documents).toBe(2); + expect(topics.types).toHaveLength(1); + expect(topics.types[0]!.multiValued).toBe(true); + expect(topics.types[0]!.distinctValues).toBe(2); + expect(valuesOf(result.keys, "topics")).toEqual([["a", 2], ["b", 1]]); + + const status = summaryOf(result.keys, "status"); + expect(status.documents).toBe(3); + expect(status.types[0]!.multiValued).toBe(false); + expect(valuesOf(result.keys, "status")).toEqual([["published", 2], ["draft", 1]]); + }); + + test("returns an empty key list when nothing has metadata", async () => { + await insertMetadataDoc("notes", {}); + + const result = listMetadata(store.db); + + expect(result).toEqual({ documents: 1, totalKeys: 0, keys: [], remainingKeys: 0 }); + }); + + test("applies the extraction gate filtered search applies", async () => { + await insertMetadataDoc("notes", { status: "visible" }); + const pendingId = await insertMetadataDoc("notes", { status: "pending" }); + const erroredId = await insertMetadataDoc("notes", { status: "errored" }); + const staleId = await insertMetadataDoc("notes", { status: "stale" }); + const inactiveId = await insertMetadataDoc("notes", { status: "inactive" }); + + store.db.prepare(`DELETE FROM document_metadata WHERE document_id = ?`).run(pendingId); + store.db.prepare(`UPDATE document_metadata SET extraction_error = 'boom' WHERE document_id = ?`).run(erroredId); + store.db.prepare(`UPDATE document_metadata SET extraction_version = ? WHERE document_id = ?`).run(METADATA_EXTRACTION_VERSION - 1, staleId); + store.db.prepare(`UPDATE documents SET active = 0 WHERE id = ?`).run(inactiveId); + + const result = listMetadata(store.db); + + // The denominator counts active documents whether or not they are extracted. + expect(result.documents).toBe(4); + expect(valuesOf(result.keys, "status")).toEqual([["visible", 1]]); + }); +}); + +describe("listMetadata scope and filter", () => { + beforeEach(async () => { + await insertMetadataDoc("notes", { status: "published", priority: 3 }); + await insertMetadataDoc("notes", { status: "draft", priority: 1 }); + await insertMetadataDoc("work", { status: "published", priority: 5 }); + await insertMetadataDoc("work", { status: "archived" }); + }); + + test("undefined collection means every collection", () => { + const result = listMetadata(store.db); + + expect(result.documents).toBe(4); + expect(summaryOf(result.keys, "status").documents).toBe(4); + expect(summaryOf(result.keys, "status").types[0]!.collections).toEqual(["notes", "work"]); + }); + + test("a single collection scopes counts and the denominator", () => { + const result = listMetadata(store.db, { collection: "notes" }); + + expect(result.documents).toBe(2); + expect(valuesOf(result.keys, "status")).toEqual([["draft", 1], ["published", 1]]); + expect(summaryOf(result.keys, "status").types[0]!.collections).toEqual(["notes"]); + }); + + test("a collection list scopes to exactly those collections", async () => { + await insertMetadataDoc("other", { status: "elsewhere" }); + + const result = listMetadata(store.db, { collection: ["notes", "work"] }); + + expect(result.documents).toBe(4); + expect(valuesOf(result.keys, "status").map(([value]) => value)).not.toContain("elsewhere"); + }); + + test("an unknown collection yields an empty scope", () => { + const result = listMetadata(store.db, { collection: "missing" }); + + expect(result).toEqual({ documents: 0, totalKeys: 0, keys: [], remainingKeys: 0 }); + }); + + test("filter narrows which documents are counted and reports how many pass", () => { + const result = listMetadata(store.db, { + filter: { field: "status", operator: "eq", value: "published" }, + }); + + expect(result.documents).toBe(4); + expect(result.filteredDocuments).toBe(2); + expect(summaryOf(result.keys, "priority").documents).toBe(2); + expect(valuesOf(result.keys, "priority")).toEqual([[3, 1], [5, 1]]); + expect(valuesOf(result.keys, "status")).toEqual([["published", 2]]); + }); + + test("filter composes with the same AST search accepts", () => { + const result = listMetadata(store.db, { + filter: { + operator: "and", + operands: [ + { field: "status", operator: "eq", value: "published" }, + { field: "priority", operator: "gte", value: 4 }, + ], + }, + }); + + expect(result.filteredDocuments).toBe(1); + expect(summaryOf(result.keys, "status").types[0]!.collections).toEqual(["work"]); + }); +}); + +describe("listMetadata match", () => { + beforeEach(async () => { + await insertMetadataDoc("notes", { topics: ["typescript", "sqlite"], owner: "docs-team", "mem-kind": "fact", priority: 3, reviewed: true }); + await insertMetadataDoc("notes", { topics: ["typescript", "PostgreSQL"], owner: "search-team", reviewers: ["docs-team", "security-team"], "mem-scope": "user", priority: 10, reviewed: false }); + }); + + function keysOf(match: MetadataFilter): string[] { + return listMetadata(store.db, { match }).keys.map(summary => summary.key); + } + + test("eq on the key field reports only that key", () => { + const result = listMetadata(store.db, { match: { field: "key", operator: "eq", value: "topics" } }); + + expect(result.keys.map(summary => summary.key)).toEqual(["topics"]); + expect(valuesOf(result.keys, "topics")).toEqual([["typescript", 2], ["PostgreSQL", 1], ["sqlite", 1]]); + }); + + test("text operators on the key field select a family of keys", () => { + expect(keysOf({ field: "key", operator: "prefix", value: "mem-" })).toEqual(["mem-kind", "mem-scope"]); + expect(keysOf({ field: "key", operator: "suffix", value: "-scope" })).toEqual(["mem-scope"]); + expect(keysOf({ field: "key", operator: "contains", value: "review" })).toEqual(["reviewed", "reviewers"]); + }); + + test("membership on the key field answers which of several names exist", () => { + expect(keysOf({ field: "key", operator: "in", value: ["tags", "topics", "labels"] })).toEqual(["topics"]); + expect(keysOf({ field: "key", operator: "nin", value: ["topics", "owner", "priority", "reviewed"] })).toEqual(["mem-kind", "mem-scope", "reviewers"]); + }); + + test("a match nothing satisfies yields an empty key list and keeps the denominator", () => { + const result = listMetadata(store.db, { match: { field: "key", operator: "prefix", value: "missing-" } }); + + expect(result.keys).toEqual([]); + expect(result.documents).toBe(2); + }); + + test("eq on the value field is a reverse lookup across keys", () => { + const result = listMetadata(store.db, { match: { field: "value", operator: "eq", value: "docs-team" } }); + + expect(result.keys.map(summary => summary.key)).toEqual(["owner", "reviewers"]); + expect(summaryOf(result.keys, "owner").documents).toBe(1); + expect(summaryOf(result.keys, "owner").types[0]!.distinctValues).toBe(1); + expect(valuesOf(result.keys, "reviewers")).toEqual([["docs-team", 1]]); + }); + + test("a key and a value condition compose with and, keeping the key's array-ness", () => { + const result = listMetadata(store.db, { + match: { + operator: "and", + operands: [ + { field: "key", operator: "eq", value: "topics" }, + { field: "value", operator: "prefix", value: "type" }, + ], + }, + }); + + expect(valuesOf(result.keys, "topics")).toEqual([["typescript", 2]]); + expect(summaryOf(result.keys, "topics").types[0]!.remainingValues).toBe(0); + // Array-ness describes the key, not the matched entries. + expect(summaryOf(result.keys, "topics").types[0]!.multiValued).toBe(true); + }); + + test("values match by their own type, never through a text form", () => { + expect(keysOf({ field: "value", operator: "eq", value: 10 })).toEqual(["priority"]); + expect(keysOf({ field: "value", operator: "eq", value: "10" })).toEqual([]); + expect(keysOf({ field: "value", operator: "eq", value: true })).toEqual(["reviewed"]); + expect(keysOf({ field: "value", operator: "contains", value: "true" })).toEqual([]); + expect(keysOf({ field: "value", operator: "prefix", value: "1" })).toEqual([]); + }); + + test("ordered comparisons on the value field select values and narrow the range", () => { + const result = listMetadata(store.db, { + match: { + operator: "and", + operands: [ + { field: "key", operator: "eq", value: "priority" }, + { field: "value", operator: "gte", value: 5 }, + ], + }, + }); + + expect(valuesOf(result.keys, "priority")).toEqual([[10, 1]]); + expect(summaryOf(result.keys, "priority").types[0]!.range).toEqual({ min: 10, median: 10, max: 10 }); + expect(summaryOf(result.keys, "priority").documents).toBe(1); + }); + + test("type on the value field selects keys by the type they hold", () => { + expect(keysOf({ field: "value", operator: "type", value: "boolean" })).toEqual(["reviewed"]); + expect(keysOf({ field: "value", operator: "type", value: "number" })).toEqual(["priority"]); + expect(keysOf({ operator: "not", operand: { field: "value", operator: "type", value: "string" } })).toEqual(["priority", "reviewed"]); + }); + + test("type on the value field reports one side of a key whose documents disagree", async () => { + await insertMetadataDoc("work", { priority: "high" }); + + const result = listMetadata(store.db, { + match: { + operator: "and", + operands: [ + { field: "key", operator: "eq", value: "priority" }, + { field: "value", operator: "type", value: "number" }, + ], + }, + }); + + const priority = summaryOf(result.keys, "priority"); + expect(priority.types.map(typeSummary => typeSummary.type)).toEqual(["number"]); + expect(priority.documents).toBe(2); + expect(priority.types[0]!.collections).toEqual(["notes"]); + }); + + test("membership and negation on the value field exclude known values", () => { + const result = listMetadata(store.db, { + match: { + operator: "and", + operands: [ + { field: "key", operator: "eq", value: "topics" }, + { operator: "not", operand: { field: "value", operator: "in", value: ["typescript"] } }, + ], + }, + }); + + expect(valuesOf(result.keys, "topics")).toEqual([["PostgreSQL", 1], ["sqlite", 1]]); + expect(keysOf({ field: "value", operator: "nin", value: ["docs-team", "search-team", "security-team", "typescript", "sqlite", "PostgreSQL", "fact", "user"] })).toEqual([]); + }); + + test("or composes across the key and value fields", () => { + const result = listMetadata(store.db, { + match: { + operator: "or", + operands: [ + { field: "key", operator: "eq", value: "owner" }, + { field: "value", operator: "eq", value: "docs-team" }, + ], + }, + }); + + expect(result.keys.map(summary => summary.key)).toEqual(["owner", "reviewers"]); + // Every owner entry satisfies the first branch, only one reviewers entry the second. + expect(valuesOf(result.keys, "owner")).toEqual([["docs-team", 1], ["search-team", 1]]); + expect(valuesOf(result.keys, "reviewers")).toEqual([["docs-team", 1]]); + }); + + test("caseInsensitive folds ASCII on either field", () => { + expect(keysOf({ field: "value", operator: "contains", value: "sql" })).toEqual(["topics"]); + expect(valuesOf(listMetadata(store.db, { match: { field: "value", operator: "contains", value: "sql" } }).keys, "topics")).toEqual([["sqlite", 1]]); + expect(valuesOf(listMetadata(store.db, { match: { field: "value", operator: "contains", value: "sql", caseInsensitive: true } }).keys, "topics")).toEqual([["PostgreSQL", 1], ["sqlite", 1]]); + expect(keysOf({ field: "key", operator: "eq", value: "TOPICS", caseInsensitive: true })).toEqual(["topics"]); + }); + + test("a number or boolean operand against the key field matches nothing", () => { + expect(keysOf({ field: "key", operator: "eq", value: 3 })).toEqual([]); + expect(keysOf({ field: "key", operator: "gte", value: 0 })).toEqual([]); + expect(keysOf({ field: "key", operator: "in", value: [true] })).toEqual([]); + // The key field is always a string, so `type string` on it is every key. + const everyKey = listMetadata(store.db).keys.map(summary => summary.key); + expect(keysOf({ field: "key", operator: "type", value: "string" })).toEqual(everyKey); + expect(keysOf({ field: "key", operator: "type", value: "number" })).toEqual([]); + }); + + test("match composes with filter: filter counts documents, match selects entries", () => { + const result = listMetadata(store.db, { + match: { field: "key", operator: "eq", value: "topics" }, + filter: { field: "reviewed", operator: "eq", value: true }, + }); + + expect(result.filteredDocuments).toBe(1); + expect(valuesOf(result.keys, "topics")).toEqual([["sqlite", 1], ["typescript", 1]]); + }); + + test("rejects conditions that have no meaning for a metadata entry", () => { + const cases: [unknown, RegExp][] = [ + [{ field: "topics", operator: "eq", value: "x" }, /^Invalid metadata match at \$: 'topics' is not a field of a metadata entry, expected 'key' or 'value'/], + [{ field: "value", operator: "exists", value: true }, /'exists' has no meaning for a single metadata entry/], + [{ field: "value", operator: "all", value: ["a"] }, /'all' has no meaning for a single metadata entry/], + [{ operator: "and", operands: [{ field: "key", operator: "eq", value: "a" }, { field: "nope", operator: "eq", value: 1 }] }, /at \$\.operands\[1\]:/], + [{ field: "value", operator: "eq", value: 3, caseInsensitive: true }, /^Invalid metadata match at \$\.caseInsensitive:/], + ]; + + for (const [match, expected] of cases) { + expect(() => listMetadata(store.db, { match: match as MetadataMatch })).toThrow(expected); + expect(() => listMetadata(store.db, { match: match as MetadataMatch })).toThrow(MetadataFilterError); + } + }); +}); + +describe("listMetadata key window", () => { + // Coverage: e 5, d 4, c 3, b 2, a 1. Report order is e, d, c, b, a. + beforeEach(async () => { + const keyNames = ["a", "b", "c", "d", "e"]; + for (const [index, key] of keyNames.entries()) { + for (let count = 0; count <= index; count += 1) await insertMetadataDoc("notes", { [key]: "x" }); + } + }); + + function keysOf(options: ListMetadataOptions = {}): string[] { + return listMetadata(store.db, options).keys.map(summary => summary.key); + } + + test("reports every key with no remainder when the vocabulary fits the window", () => { + const result = listMetadata(store.db); + + expect(result.keys.map(summary => summary.key)).toEqual(["e", "d", "c", "b", "a"]); + expect(result.totalKeys).toBe(5); + expect(result.remainingKeys).toBe(0); + }); + + test("keyLimit windows keys in report order and reports the exact remainder", () => { + const result = listMetadata(store.db, { keyLimit: 2 }); + + expect(result.keys.map(summary => summary.key)).toEqual(["e", "d"]); + expect(result.totalKeys).toBe(5); + expect(result.remainingKeys).toBe(3); + }); + + test("keyOffset pages through the vocabulary without gaps or repeats", () => { + const pages = [0, 2, 4].map(keyOffset => listMetadata(store.db, { keyLimit: 2, keyOffset })); + + expect(pages.map(page => page.keys.map(summary => summary.key))).toEqual([["e", "d"], ["c", "b"], ["a"]]); + expect(pages.map(page => page.remainingKeys)).toEqual([3, 1, 0]); + expect(pages.every(page => page.totalKeys === 5)).toBe(true); + }); + + test("an offset past the end reports no keys and keeps the total", () => { + const result = listMetadata(store.db, { keyOffset: 10 }); + + expect(result.keys).toEqual([]); + expect(result.totalKeys).toBe(5); + expect(result.remainingKeys).toBe(0); + }); + + test("an infinite keyLimit removes the window", () => { + expect(keysOf({ keyLimit: Infinity })).toEqual(["e", "d", "c", "b", "a"]); + }); + + test("the window applies to the keys the match admits", () => { + const result = listMetadata(store.db, { match: { field: "key", operator: "in", value: ["a", "c", "e"] }, keyLimit: 2 }); + + expect(result.keys.map(summary => summary.key)).toEqual(["e", "c"]); + expect(result.totalKeys).toBe(3); + expect(result.remainingKeys).toBe(1); + }); + + test("a key summary inside the window is complete", async () => { + await insertMetadataDoc("work", { e: ["y", "z"], d: 1 }); + + const result = listMetadata(store.db, { keyLimit: 1 }); + const e = summaryOf(result.keys, "e"); + + expect(e.documents).toBe(6); + expect(e.types[0]!.multiValued).toBe(true); + expect(e.types[0]!.distinctValues).toBe(3); + expect(e.types[0]!.collections).toEqual(["notes", "work"]); + expect(result.keys).toHaveLength(1); + }); + + test("collection provenance uses UTF-8 ordering without SQLite 3.44 aggregate syntax", async () => { + for (const collection of ["𐀀", "", "notes", "B"]) { + await insertMetadataDoc(collection, { provenance: "shared" }); + } + const observed = observeStatements(store.db, sql => { + expect(sql).not.toMatch(/json_group_array\([^)]*\bORDER\s+BY/iu); + }); + const report = listMetadata(observed, { match: { field: "key", operator: "eq", value: "provenance" } }); + expect(summaryOf(report.keys, "provenance").types[0]!.collections).toEqual(["B", "notes", "", "𐀀"]); + }); + + test("pages are prefixes of the full result under one collation, on both readers", async () => { + // Equal coverage, so order falls to the key name. BINARY collation puts + // uppercase before "_", "_" before lowercase, and lowercase before "é". + await insertMetadataDoc("work", { _id: 1, a: 1, B: 1, "é": 1, Z: 1 }); + const names = ["B", "Z", "_id", "a", "é"]; + + const fullResult = listMetadata(store.db, { collection: "work" }).keys.map(summary => summary.key); + const pages = [0, 2, 4].flatMap(keyOffset => + listMetadata(store.db, { collection: "work", keyLimit: 2, keyOffset }).keys.map(summary => summary.key)); + + expect(fullResult).toEqual(names); + expect(pages).toEqual(names); + expect(listMetadataKeys(store.db, ["work"]).map(overview => overview.key)).toEqual(names); + expect(listMetadataKeys(store.db, ["work"], 3).map(overview => overview.key)).toEqual(names.slice(0, 3)); + }); + + test("ranks keys once per report and groups values once for both totals and windows", () => { + const statements: string[] = []; + const observed = observeStatements(store.db, sql => statements.push(sql)); + const options: ListMetadataOptions[] = [ + { keyLimit: 5, valueLimit: 3 }, + { collection: "notes", keyLimit: 5, valueLimit: 3 }, + { filter: { field: "e", operator: "prefix", value: "x" }, keyLimit: 5 }, + {}, + ]; + for (const statistics of [false, true]) { + if (statistics) store.db.exec("ANALYZE"); + for (const option of options) { + statements.length = 0; + listMetadata(observed, option); + expect(statements.filter(sql => sql.includes("selected_keys AS"))).toHaveLength(1); + expect(statements.filter(sql => sql.includes("GROUP BY mv.key, mv.value_type, mv.text_value"))).toHaveLength(1); + if (option.filter) expect(statements.some(sql => sql.includes("d.id IN (SELECT mv.document_id"))).toBe(true); + for (const sql of statements) { + const plan = planOf(sql); + expect(plan.some(step => step.includes(" EXISTS ")), sql).toBe(false); + if (sql.includes("d.id IN (SELECT mv.document_id")) expect(plan.some(step => step.includes("LIST SUBQUERY")), sql).toBe(true); + if (!sql.includes("selected_keys AS")) continue; + expect(plan.some(step => step.includes("MATERIALIZE selected_keys")), sql).toBe(true); + expect(plan.some(step => step.includes("CO-ROUTINE selected_keys")), sql).toBe(false); + } + } + statements.length = 0; + listMetadataKeys(observed, undefined, 10); + expect(statements).toHaveLength(1); + expect(statements[0]).toContain("mv.ordinal = 0"); + expect(statements[0]).not.toContain("number_value"); + } + }); +}); + +describe("metadata overviews", () => { + test("batched collection overviews agree with full reports through lifecycle changes", async () => { + await insertMetadataDoc("notes", { tags: ["a", "b", "c"], priority: 1 }); + const changed = await insertMetadataDoc("notes", { tags: 7, priority: "high" }); + await insertMetadataDoc("work", { tags: true, "é": "x", "😀": "x" }); + const stale = await insertMetadataDoc("work", { old: "hidden" }); + const inactive = await insertMetadataDoc("work", { deleted: "hidden" }); + store.db.prepare("UPDATE document_metadata SET extraction_version = 0 WHERE document_id = ?").run(stale); + store.db.prepare("UPDATE documents SET active = 0 WHERE id = ?").run(inactive); + + for (const phase of ["initial", "replaced", "renamed", "deleted"] as const) { + if (phase === "replaced") replaceDocumentMetadata(store.db, changed, { metadata: { tags: [true, false] }, extractionVersion: METADATA_EXTRACTION_VERSION }); + if (phase === "renamed") store.db.prepare("UPDATE documents SET collection = 'renamed' WHERE collection = 'notes'").run(); + if (phase === "deleted") store.db.prepare("DELETE FROM documents WHERE id = ?").run(changed); + for (const keyLimit of [1, 10, Infinity]) { + const statements: string[] = []; + const observed = observeStatements(store.db, sql => statements.push(sql)); + const overviews = listMetadataCollectionSummaries(observed, keyLimit); + expect(statements).toHaveLength(1); + for (const collection of ["notes", "renamed", "work", "missing"]) { + const report = listMetadata(store.db, { collection, keyLimit }); + const expected = { totalKeys: report.totalKeys, keys: report.keys.map(key => ({ key: key.key, documents: key.documents, types: key.types.map(type => type.type) })) }; + expect(overviews.get(collection) ?? { totalKeys: 0, keys: [] }).toEqual(expected); + expect(getMetadataOverview(store.db, [collection], keyLimit)).toEqual(expected); + } + } + } + }); +}); + +describe("entry text predicates", () => { + test("positive, negated, and compound matches partition empty and nonempty strings", async () => { + const values = ["", "draft", "PUBLISHED", "abc\u0000XYZ", "\u0000", "😀XYZ", 7, true]; + for (const value of values) await insertMetadataDoc("notes", { label: value }); + + for (const operator of ["contains", "prefix", "suffix"] as const) { + for (const operand of ["ft", "PUB", "XYZ", "\u0000", "😀", "nomatch"]) { + for (const caseInsensitive of [false, true]) { + const fold = (text: string) => caseInsensitive ? text.replace(/[A-Z]/gu, letter => letter.toLowerCase()) : text; + const expected = values.map(value => { + if (typeof value !== "string") return false; + const text = fold(value); + const needle = fold(operand); + return operator === "contains" ? text.includes(needle) : operator === "prefix" ? text.startsWith(needle) : text.endsWith(needle); + }); + const condition = { field: "value", operator, value: operand, caseInsensitive } as const; + const predicates: MetadataMatch[] = [ + condition, + { operator: "not", operand: condition }, + { operator: "and", operands: [{ field: "key", operator: "eq", value: "label" }, { operator: "not", operand: condition }] }, + { operator: "or", operands: [condition, { operator: "not", operand: condition }] }, + ]; + const expectations = [expected, expected.map(value => !value), expected.map(value => !value), values.map(() => true)]; + for (const [index, predicate] of predicates.entries()) { + const compiled = compileMetadataMatch(predicate, "mv"); + const rows = store.db.prepare(`SELECT ${compiled.sql} AS matched FROM document_metadata_values mv ORDER BY document_id`).all(...compiled.params) as { matched: number }[]; + expect(rows.map(row => row.matched)).toEqual(expectations[index]!.map(Number)); + const report = listMetadata(store.db, { match: predicate, valueLimit: Infinity }); + expect(report.keys[0]?.documents ?? 0).toBe(expectations[index]!.filter(Boolean).length); + } + + const filter: MetadataFilter = { operator: "not", operand: { ...condition, field: "label" } }; + for (const scope of ["candidates", "corpus"] as const) { + const compiled = compileMetadataFilter(filter, "d", scope); + const rows = store.db.prepare(`SELECT ${compiled.sql} AS matched FROM documents d ORDER BY id`).all(...compiled.params) as { matched: number }[]; + expect(rows.map(row => row.matched)).toEqual(expected.map(value => Number(!value))); + } + } + } + } + }); +}); + +describe("listMetadata consistency", () => { + test("reads the whole report from one snapshot while another connection writes", async () => { + const documentId = await insertMetadataDoc("notes", { priority: 1 }); + const writer = createStore(store.db.prepare("PRAGMA database_list").all().map(row => (row as { file: string }).file)[0]!); + + try { + let rewritten = false; + const observed = observeStatements(store.db, sql => { + // Commit a change from the second connection after the type statistics + // (min and max) have been read and before the medians and values are, + // the interleaving that would report a range no document ever had. + if (rewritten || !sql.includes("PARTITION BY mv.key ORDER BY mv.number_value")) return; + rewritten = true; + replaceDocumentMetadata(writer.db, documentId, { metadata: { priority: 100 }, extractionVersion: METADATA_EXTRACTION_VERSION }); + }); + + const priority = summaryOf(listMetadata(observed).keys, "priority").types[0]!; + + expect(rewritten).toBe(true); + expect(priority.range).toEqual({ min: 1, median: 1, max: 1 }); + expect(priority.values).toEqual([{ value: 1, documents: 1 }]); + expect(summaryOf(listMetadata(store.db).keys, "priority").types[0]!.values).toEqual([{ value: 100, documents: 1 }]); + } finally { + writer.close(); + } + }); +}); + +describe("listMetadata options", () => { + test("rejects a window or ordering option outside its domain", () => { + const cases: [ListMetadataOptions, RegExp][] = [ + [{ keyLimit: 0 }, /^Invalid keyLimit: expected a positive integer or Infinity, received 0$/], + [{ keyLimit: 1.5 }, /^Invalid keyLimit: expected a positive integer or Infinity, received 1.5$/], + [{ keyOffset: -1 }, /^Invalid keyOffset: expected a non-negative integer, received -1$/], + [{ valueLimit: -Infinity }, /^Invalid valueLimit: expected a positive integer or Infinity, received -Infinity$/], + [{ valueOffset: 0.5 }, /^Invalid valueOffset: expected a non-negative integer, received 0.5$/], + [{ minCount: 0 }, /^Invalid minCount: expected a positive integer, received 0$/], + [{ sort: "size" as "count" }, /^Invalid sort: expected 'count' or 'value', received "size"$/], + ]; + + for (const [options, expected] of cases) { + expect(() => listMetadata(store.db, options)).toThrow(expected); + expect(() => listMetadata(store.db, options)).toThrow(MetadataOptionError); + } + }); + + test("names the offending option on the error", () => { + try { + listMetadata(store.db, { keyOffset: -1 }); + throw new Error("expected listMetadata to throw"); + } catch (error) { + expect(error).toBeInstanceOf(MetadataOptionError); + expect((error as MetadataOptionError).option).toBe("keyOffset"); + } + }); + + test("rejects integers SQLite cannot bind as INTEGER, and binds the largest it can", async () => { + await insertMetadataDoc("notes", { status: "x" }); + const largest = Number.MAX_SAFE_INTEGER; + + expect(() => listMetadata(store.db, { keyLimit: 1e30 })).toThrow(MetadataOptionError); + expect(() => listMetadata(store.db, { keyOffset: 2 ** 53 })).toThrow(/^Invalid keyOffset: expected a non-negative integer, received 9007199254740992$/); + expect(() => listMetadata(store.db, { valueOffset: 1e30 })).toThrow(MetadataOptionError); + expect(() => listMetadata(store.db, { minCount: 1e30 })).toThrow(MetadataOptionError); + + expect(listMetadata(store.db, { keyLimit: largest, keyOffset: largest }).keys).toEqual([]); + expect(summaryOf(listMetadata(store.db, { valueLimit: largest, valueOffset: largest }).keys, "status").types[0]!.values).toEqual([]); + expect(summaryOf(listMetadata(store.db, { valueLimit: largest }).keys, "status").types[0]!.values).toHaveLength(1); + }); + + test("the production value window includes its exact integer endpoint past 2^53", async () => { + await insertMetadataDoc("notes", { label: "x" }); + let windowChecked = false; + const observed = observeStatements(store.db, () => {}, (sql, params) => { + const windowSql = /\((rank > \? AND rank <= .*)\) AS in_window/u.exec(sql)?.[1]; + if (!windowSql) return; + windowChecked = true; + + // Execute the actual emitted predicate and bindings on a tiny exact-rank + // fixture. Reading the rank itself as a JS number would round it again. + const rows = store.db.prepare(` + WITH ranks(rank) AS (VALUES (9007199254740991), (9007199254740992), (9007199254740993), (9007199254740994)) + SELECT CAST(rank AS TEXT) AS rank FROM ranks WHERE ${windowSql} ORDER BY rank + `).all(...params.slice(-3)); + expect(rows).toEqual([{ rank: "9007199254740992" }, { rank: "9007199254740993" }]); + }); + + listMetadata(observed, { valueOffset: Number.MAX_SAFE_INTEGER, valueLimit: 2 }); + expect(windowChecked).toBe(true); + }); + + test("rejects a filter and match whose bindings together exceed the statement budget", async () => { + await insertMetadataDoc("notes", { status: "x" }); + const members = Array.from({ length: 64 }, (_, index) => `v${index}`); + // 7 groups of 31 conditions: 225 nodes, inside every parser limit. + const wideFilter: MetadataFilter = { + operator: "and", + operands: Array.from({ length: 7 }, () => ({ + operator: "and" as const, + operands: Array.from({ length: 31 }, () => ({ field: "status", operator: "all" as const, value: members })), + })), + }; + const wideMatch: MetadataMatch = { + operator: "or", + operands: Array.from({ length: 7 }, () => ({ + operator: "or" as const, + operands: Array.from({ length: 31 }, () => ({ field: "value" as const, operator: "in" as const, value: members })), + })), + }; + + // 217 conditions binding field and value per member: 27,776, under budget alone. + expect(listMetadata(store.db, { filter: wideFilter }).totalKeys).toBe(0); + expect(listMetadata(store.db, { match: wideMatch }).totalKeys).toBe(0); + + try { + listMetadata(store.db, { filter: wideFilter, match: wideMatch }); + throw new Error("expected listMetadata to throw"); + } catch (error) { + expect(error).toBeInstanceOf(MetadataBindingBudgetError); + const budgetError = error as MetadataBindingBudgetError; + expect(budgetError.budget).toBe(METADATA_SQL_BINDING_BUDGET); + expect(budgetError.bindings).toBeGreaterThan(METADATA_SQL_BINDING_BUDGET); + expect(budgetError.message).toMatch(/^filter and match together bind \d+ SQL parameters, over the budget of 30000\. Shorten their membership lists\.$/); + } + }); +}); + +describe("listMetadata value window", () => { + beforeEach(async () => { + const tags = ["a", "a", "a", "b", "b", "c", "d", "d", "d", "d"]; + for (const tag of tags) await insertMetadataDoc("notes", { tag }); + }); + + test("defaults to count order with a stable tiebreak", () => { + expect(valuesOf(listMetadata(store.db).keys, "tag")).toEqual([["d", 4], ["a", 3], ["b", 2], ["c", 1]]); + }); + + test("sort by value orders ascending", () => { + expect(valuesOf(listMetadata(store.db, { sort: "value" }).keys, "tag")).toEqual([["a", 3], ["b", 2], ["c", 1], ["d", 4]]); + }); + + test("valueLimit windows values and reports the exact remainder", () => { + const tag = summaryOf(listMetadata(store.db, { valueLimit: 2 }).keys, "tag").types[0]!; + + expect(tag.values.map(count => count.value)).toEqual(["d", "a"]); + expect(tag.distinctValues).toBe(4); + expect(tag.remainingValues).toBe(2); + }); + + test("an infinite valueLimit removes the window", () => { + const tag = summaryOf(listMetadata(store.db, { valueLimit: Infinity }).keys, "tag").types[0]!; + + expect(tag.values).toHaveLength(4); + expect(tag.remainingValues).toBe(0); + }); + + test("valueOffset pages through the values without gaps or repeats", () => { + const pages = [0, 2, 4].map(valueOffset => + summaryOf(listMetadata(store.db, { valueLimit: 2, valueOffset }).keys, "tag").types[0]!); + + expect(pages.map(tag => tag.values.map(count => count.value))).toEqual([["d", "a"], ["b", "c"], []]); + expect(pages.map(tag => tag.remainingValues)).toEqual([2, 0, 0]); + expect(pages.every(tag => tag.distinctValues === 4)).toBe(true); + }); + + test("valueOffset follows the requested sort", () => { + const tag = summaryOf(listMetadata(store.db, { sort: "value", valueLimit: 2, valueOffset: 1 }).keys, "tag").types[0]!; + + expect(tag.values.map(count => count.value)).toEqual(["b", "c"]); + }); + + test("minCount drops the tail from the values, distinct count, and remainder", () => { + const tag = summaryOf(listMetadata(store.db, { minCount: 3, valueLimit: 1 }).keys, "tag").types[0]!; + + expect(tag.values).toEqual([{ value: "d", documents: 4 }]); + expect(tag.distinctValues).toBe(2); + expect(tag.remainingValues).toBe(1); + // Coverage still counts every document holding the key. + expect(tag.documents).toBe(10); + }); + + test("a minCount nothing meets leaves the key with an empty window", () => { + const tag = summaryOf(listMetadata(store.db, { minCount: 99 }).keys, "tag").types[0]!; + + expect(tag.values).toEqual([]); + expect(tag.distinctValues).toBe(0); + expect(tag.remainingValues).toBe(0); + }); +}); + +describe("listMetadata numbers", () => { + test("reports min, median, and max for an odd count", async () => { + for (const priority of [5, 1, 3]) await insertMetadataDoc("notes", { priority }); + + const priority = summaryOf(listMetadata(store.db).keys, "priority").types[0]!; + + expect(priority.range).toEqual({ min: 1, median: 3, max: 5 }); + expect(priority.values.map(count => count.value)).toEqual([1, 3, 5]); + }); + + test("averages the middle values for an even count", async () => { + for (const priority of [1, 2, 3, 10]) await insertMetadataDoc("notes", { priority }); + + expect(summaryOf(listMetadata(store.db).keys, "priority").types[0]!.range).toEqual({ min: 1, median: 2.5, max: 10 }); + }); + + test("the even-count median is finite and correctly rounded across the double range", async () => { + const cases: [number, number, number][] = [ + [1e308, 1.5e308, 1.25e308], + [-1.5e308, -1e308, -1.25e308], + [-1e308, 1e308, 0], + [Number.MIN_VALUE, 2 * Number.MIN_VALUE, 2 * Number.MIN_VALUE], + [-2 * Number.MIN_VALUE, -Number.MIN_VALUE, -2 * Number.MIN_VALUE], + [Number.MAX_VALUE, Number.MAX_VALUE, Number.MAX_VALUE], + ]; + + for (const [index, [lower, upper]] of cases.entries()) { + await insertMetadataDoc(`c${index}`, { n: lower }); + await insertMetadataDoc(`c${index}`, { n: upper }); + } + + for (const [index, [lower, upper, median]] of cases.entries()) { + const n = summaryOf(listMetadata(store.db, { collection: `c${index}` }).keys, "n").types[0]!; + expect(n.range, `${lower}, ${upper}`).toEqual({ min: lower, median, max: upper }); + } + }); + + test("median counts every value row, including array elements", async () => { + await insertMetadataDoc("notes", { scores: [1, 1, 1] }); + await insertMetadataDoc("notes", { scores: [9] }); + + const scores = summaryOf(listMetadata(store.db).keys, "scores").types[0]!; + + expect(scores.range).toEqual({ min: 1, median: 1, max: 9 }); + expect(scores.values).toEqual([{ value: 1, documents: 1 }, { value: 9, documents: 1 }]); + }); + + test("strings and booleans carry no range", async () => { + await insertMetadataDoc("notes", { status: "x", reviewed: true }); + + const result = listMetadata(store.db); + + expect(summaryOf(result.keys, "status").types[0]!.range).toBeUndefined(); + expect(summaryOf(result.keys, "reviewed").types[0]!.range).toBeUndefined(); + }); +}); + +describe("listMetadata type conflicts and attribution", () => { + test("splits a key by type and partitions its documents", async () => { + await insertMetadataDoc("notes", { priority: 3 }); + await insertMetadataDoc("notes", { priority: 1 }); + await insertMetadataDoc("work", { priority: "high" }); + + const priority = summaryOf(listMetadata(store.db).keys, "priority"); + + expect(priority.documents).toBe(3); + expect(priority.types.map(typeSummary => [typeSummary.type, typeSummary.documents])).toEqual([["number", 2], ["string", 1]]); + expect(priority.types[0]!.range).toEqual({ min: 1, median: 2, max: 3 }); + expect(priority.types[0]!.collections).toEqual(["notes"]); + expect(priority.types[1]!.collections).toEqual(["work"]); + expect(priority.types[1]!.values).toEqual([{ value: "high", documents: 1 }]); + }); + + test("orders equally covered types by name", async () => { + await insertMetadataDoc("notes", { flag: true }); + await insertMetadataDoc("notes", { flag: "yes" }); + + expect(summaryOf(listMetadata(store.db).keys, "flag").types.map(typeSummary => typeSummary.type)).toEqual(["boolean", "string"]); + }); + + test("orders keys by coverage descending, then name", async () => { + await insertMetadataDoc("notes", { zeta: 1, alpha: 1, mid: 1 }); + await insertMetadataDoc("notes", { zeta: 1, alpha: 1 }); + await insertMetadataDoc("notes", { zeta: 1 }); + + expect(listMetadata(store.db).keys.map(summary => summary.key)).toEqual(["zeta", "alpha", "mid"]); + }); +}); + +describe("listMetadata agrees with filtered search", () => { + test("every reported value is reachable through an eq filter under the same scope", async () => { + await insertMetadataDoc("notes", { topics: ["a", "b"], priority: 3, reviewed: true, owner: "docs-team" }); + await insertMetadataDoc("notes", { topics: ["b"], priority: 1.5, reviewed: false }); + await insertMetadataDoc("work", { topics: ["c"], priority: 3 }); + const pendingId = await insertMetadataDoc("notes", { topics: ["ghost"] }); + store.db.prepare(`DELETE FROM document_metadata WHERE document_id = ?`).run(pendingId); + + const scopes: ListMetadataOptions[] = [{}, { collection: "notes" }, { collection: ["notes", "work"] }]; + for (const scope of scopes) { + const result = listMetadata(store.db, { ...scope, valueLimit: Infinity }); + expect(result.keys.length).toBeGreaterThan(0); + + for (const summary of result.keys) { + for (const typeSummary of summary.types) { + for (const count of typeSummary.values) { + const hits = searchFTS(store.db, "doc", 100, scope.collection, { field: summary.key, operator: "eq", value: count.value }); + expect(hits, `${summary.key} = ${String(count.value)}`).toHaveLength(count.documents); + } + } + } + } + }); +}); diff --git a/test/metadata-filter.test-d.ts b/test/metadata-filter.test-d.ts new file mode 100644 index 000000000..9f6cb4e76 --- /dev/null +++ b/test/metadata-filter.test-d.ts @@ -0,0 +1,74 @@ +/** + * metadata-filter.test-d.ts - Compile-time shape of the predicate grammar. + * MetadataFilter and MetadataMatch share one recursive grammar and differ in + * the conditions they admit. Checked by vitest's typecheck pass. + */ + +import { describe, test, expectTypeOf } from "vitest"; +import { + compileMetadataMatch, + parseMetadataFilter, + parseMetadataMatch, + type MetadataCondition, + type MetadataEntryCondition, + type MetadataFilter, + type MetadataMatch, + type MetadataPredicate, +} from "../src/metadata-filter.js"; +import type { ListMetadataOptions } from "../src/metadata-store.js"; + +describe("MetadataPredicate", () => { + test("filter and match are the one grammar over different conditions", () => { + expectTypeOf().toEqualTypeOf>(); + expectTypeOf().toEqualTypeOf>(); + expectTypeOf(parseMetadataFilter).returns.toEqualTypeOf(); + expectTypeOf(parseMetadataMatch).returns.toEqualTypeOf(); + expectTypeOf(compileMetadataMatch).parameter(0).toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + }); + + test("a match composes with and, or, and not like a filter", () => { + const match: MetadataMatch = { + operator: "and", + operands: [ + { field: "key", operator: "prefix", value: "mem-" }, + { operator: "not", operand: { field: "value", operator: "type", value: "boolean" } }, + { operator: "or", operands: [ + { field: "value", operator: "gte", value: 3 }, + { field: "value", operator: "in", value: ["a", "b"], caseInsensitive: true }, + ] }, + ], + }; + expectTypeOf(match).toMatchTypeOf(); + }); + + test("a match names only the fields of a metadata entry", () => { + // @ts-expect-error a document's metadata key is not a field of an entry + const byMetadataKey: MetadataMatch = { field: "topics", operator: "eq", value: "typescript" }; + // @ts-expect-error nested in a group as well + const nested: MetadataMatch = { operator: "and", operands: [{ field: "topics", operator: "eq", value: "x" }] }; + expectTypeOf(byMetadataKey).toEqualTypeOf(); + expectTypeOf(nested).toEqualTypeOf(); + }); + + test("a match admits no condition about the set of values a document holds", () => { + // @ts-expect-error `exists` has no meaning for a single entry + const exists: MetadataMatch = { field: "value", operator: "exists", value: true }; + // @ts-expect-error `all` has no meaning for a single entry + const all: MetadataMatch = { field: "value", operator: "all", value: ["a"] }; + expectTypeOf(exists).toEqualTypeOf(); + expectTypeOf(all).toEqualTypeOf(); + }); + + test("a filter still admits every document condition", () => { + const filter: MetadataFilter = { + operator: "and", + operands: [ + { field: "topics", operator: "all", value: ["typescript", "sqlite"] }, + { field: "reviewed", operator: "exists", value: true }, + { field: "key", operator: "eq", value: "a metadata key may be named key" }, + ], + }; + expectTypeOf(filter).toMatchTypeOf(); + }); +}); diff --git a/test/metadata-filter.test.ts b/test/metadata-filter.test.ts index 7510fd2ee..7df69e7dc 100644 --- a/test/metadata-filter.test.ts +++ b/test/metadata-filter.test.ts @@ -8,7 +8,10 @@ import { openDatabase } from "../src/db.js"; import type { Database } from "../src/db.js"; import { parseMetadataFilter, + parseMetadataMatch, compileMetadataFilter, + type FilterScope, + compileMetadataMatch, MetadataFilterError, METADATA_FILTER_LIMITS, type MetadataFilter, @@ -159,6 +162,60 @@ describe("parseMetadataFilter", () => { }); }); +describe("parseMetadataMatch", () => { + test("accepts the filter grammar with 'key' and 'value' as the condition fields", () => { + const match = { + operator: "and", + operands: [ + { field: "key", operator: "prefix", value: "mem-" }, + { operator: "or", operands: [ + { field: "value", operator: "gte", value: 3 }, + { field: "value", operator: "type", value: "boolean" }, + { operator: "not", operand: { field: "value", operator: "in", value: ["Draft"], caseInsensitive: true } }, + ] }, + ], + }; + expect(parseMetadataMatch(match)).toEqual(match); + }); + + test("rejects other fields and the two set operators, naming the match", () => { + const cases: [unknown, RegExp][] = [ + [{ field: "topics", operator: "eq", value: "x" }, /^Invalid metadata match at \$: 'topics' is not a field of a metadata entry, expected 'key' or 'value'$/], + [{ field: "value", operator: "exists", value: true }, /^Invalid metadata match at \$: 'exists' has no meaning for a single metadata entry$/], + [{ field: "key", operator: "all", value: ["a"] }, /'all' has no meaning for a single metadata entry/], + [{ operator: "not", operand: { field: "value", operator: "eq" } }, /^Invalid metadata match at \$\.operand: 'eq' requires a 'value'$/], + ["nope", /^Invalid metadata match at \$: each filter node must be an object$/], + ]; + for (const [input, expected] of cases) { + expect(() => parseMetadataMatch(input)).toThrow(expected); + expect(() => parseMetadataMatch(input)).toThrow(MetadataFilterError); + } + // The filter keeps its own name. + expect(() => parseMetadataFilter("nope")).toThrow(/^Invalid metadata filter at \$:/); + }); +}); + +describe("compileMetadataMatch", () => { + test("compiles to a predicate over the row alias with every operand bound", () => { + const compiled = compileMetadataMatch(parseMetadataMatch({ + operator: "and", + operands: [ + { field: "key", operator: "eq", value: "k'; --" }, + { field: "value", operator: "suffix", value: "V'; --", caseInsensitive: true }, + { field: "value", operator: "nin", value: [1, 2] }, + ], + }), "row"); + + expect(compiled.sql).not.toContain("'; --"); + expect(compiled.sql).not.toContain("EXISTS"); + expect(compiled.sql).toContain("row.key = ?"); + expect(compiled.sql).toContain("lower(row.text_value)"); + expect(compiled.sql).toContain("row.number_value NOT IN (?, ?)"); + // The suffix binds its operand's UTF-8 byte length ahead of the operand. + expect(compiled.params).toEqual(["k'; --", 6, "v'; --", 1, 2]); + }); +}); + // ============================================================================= // SQL compilation and semantics // ============================================================================= @@ -189,8 +246,8 @@ describe("compileMetadataFilter semantics", () => { return documentId; } - function matchPaths(filter: MetadataFilter): string[] { - const compiled = compileMetadataFilter(parseMetadataFilter(filter), "d"); + function matchPathsIn(filter: MetadataFilter, scope: FilterScope): string[] { + const compiled = compileMetadataFilter(parseMetadataFilter(filter), "d", scope); const rows = db.prepare(` SELECT d.path FROM documents d JOIN document_metadata dm ON dm.document_id = d.id @@ -202,6 +259,13 @@ describe("compileMetadataFilter semantics", () => { return rows.map(row => row.path); } + /** Both scopes are one semantics in two SQL forms, so every case checks they agree. */ + function matchPaths(filter: MetadataFilter): string[] { + const candidates = matchPathsIn(filter, "candidates"); + expect(matchPathsIn(filter, "corpus"), `corpus scope for ${JSON.stringify(filter)}`).toEqual(candidates); + return candidates; + } + test("eq matches each scalar type exactly, without coercion", () => { insertDoc("str.md", { status: "published" }); insertDoc("num.md", { status: 1 }); @@ -451,6 +515,36 @@ describe("compileMetadataFilter semantics", () => { } }); + test("the corpus scope compiles each condition to one uncorrelated document set", () => { + for (let index = 0; index < 64; index++) { + insertDoc(`d${index}.md`, { status: index % 2 ? "published" : "draft", priority: index }); + } + const filter = parseMetadataFilter({ + operator: "and", + operands: [ + { field: "priority", operator: "gt", value: 3 }, + { operator: "not", operand: { field: "status", operator: "exists", value: false } }, + ], + }); + const compiled = compileMetadataFilter(filter, "d", "corpus"); + expect(compiled.sql).toBe( + "(d.id IN (SELECT mv.document_id FROM document_metadata_values mv WHERE mv.key = ? AND mv.value_type = 'number' AND mv.number_value > ?)" + + " AND NOT NOT d.id IN (SELECT mv.document_id FROM document_metadata_values mv WHERE mv.key = ?))", + ); + expect(compiled.params).toEqual(["priority", 3, "status"]); + + for (const statistics of [false, true]) { + if (statistics) db.exec("ANALYZE"); + const plan = (db.prepare(`EXPLAIN QUERY PLAN SELECT COUNT(*) FROM documents d WHERE ${compiled.sql}`) + .all(...compiled.params) as { detail: string }[]) + .map(row => row.detail); + // The document sets are built once (LIST SUBQUERY), by covering index, and probed per document. + expect(plan.filter(step => step.includes("LIST SUBQUERY"))).toHaveLength(2); + expect(plan.some(step => step.includes(" EXISTS ")), plan.join(" | ")).toBe(false); + expect(plan.some(step => step.includes("idx_metadata_number_lookup (key=? AND number_value>?)")), plan.join(" | ")).toBe(true); + } + }); + test("compiled SQL never interpolates user keys or values", () => { const filter = parseMetadataFilter({ operator: "and", diff --git a/vitest.config.ts b/vitest.config.ts index 463e72385..415fd4db3 100644 --- a/vitest.config.ts +++ b/vitest.config.ts @@ -5,5 +5,10 @@ export default defineConfig({ testTimeout: 30000, fileParallelism: false, include: ["test/**/*.test.ts"], + typecheck: { + enabled: true, + include: ["test/**/*.test-d.ts"], + ignoreSourceErrors: true, + }, }, }); From 68f62bbeb5ac927159ee4bed4b429a04284ba77b Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Sat, 12 Sep 2026 13:17:23 -0700 Subject: [PATCH 18/82] feat(metadata): add qmd collection metadata drill-down Adds the CLI surface for metadata discovery so an agent can learn what to filter on before it queries: - `qmd collection metadata [name...]` prints one block per key: a header with type, coverage, and distinct count, then a value window. Strings list vertically with right-aligned counts, numbers show min/median/max plus the values when they fit, booleans print true/false counts. Keys whose documents disagree on type split per type, with contributing collections in the multi-collection view. - `--filter ` counts only matching documents. The output then opens with a `filter:` line stating how many documents pass, and every key's coverage is measured against that population, so the numbers an agent reads are the numbers a filtered search would see. - `--match ` selects which metadata entries are reported, with the same AST as `--filter` evaluated against each entry (a condition's `field` is the entry's `key` or `value`). Both JSON flags share one parser and report malformed JSON and invalid nodes the same way. - Two windows, named the same on every surface: `--key-limit ` (default 50) with `--key-offset ` and `--all-keys`, and `--value-limit ` (default 10) with `--value-offset ` and `--all-values`. Every windowed list ends with the exact remainder and the flags that reach it, and an offset past the last key says how many keys there are. `-n` and `--all` are the single-window search flags and exit with a pointer at the discovery ones. `--sort count|value` orders values, `--min-count` drops the tail. Omitting the name covers the default collections, as an unscoped search does. - src/metadata-format.ts holds the renderer so the MCP tool can print the same shape. It takes the CLI's color palette or none. The empty states render there too, so the filter line survives a filter whose documents declare no metadata. A string value prints bare only when the bare form is unambiguous. One that is empty, padded, contains a quote, a backslash, a control character, a line separator, or a delimiter of the compact `value (count), ...` list a type split prints, or that reads as a JSON number, boolean, or null, or as the list's remainder tail, prints as a JSON string with every control character escaped. One rule for both layouts, so a value never prints two ways, and `"a (1), b" (1)` cannot be read as two values. Ordinary values are unchanged, so the common case costs nothing. - A filter and match that together exceed the SQL binding budget exit with the store's message. - The pending-extraction warning prints on stderr whenever documents are gated out, since discovery always applies the gate. The renderer suite reads the compact list back the way a reader must, honoring quotes, and checks every item round-trips. Covers the keys view, both windows with paging and their footers, the filter line and denominator (including over an empty result), reverse lookup, a composed match across both fields, type selection, sort and min-count, default scope, empty matches, and exit codes for unknown collections and invalid flags, filters, and matches. The renderer has its own suite for string identity (round trip through JSON.parse, no raw control characters) and the empty states. Numeric flags reject unsafe integers as usage errors before opening the index, preserving the original input in the message. Unsupported format flags fail explicitly with a pointer to the structured surfaces, rather than silently returning text to a caller expecting JSON. Assisted-by: Claude Fable 5.1 via Pi --- src/cli/qmd.ts | 202 ++++++++++++++++++++- src/metadata-format.ts | 226 ++++++++++++++++++++++++ test/metadata-cli.test.ts | 331 +++++++++++++++++++++++++++++++++++ test/metadata-format.test.ts | 139 +++++++++++++++ 4 files changed, 889 insertions(+), 9 deletions(-) create mode 100644 src/metadata-format.ts create mode 100644 test/metadata-format.test.ts diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 7e386b954..ed9590658 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -86,9 +86,17 @@ import { type ReindexResult, type ChunkStrategy, } from "../store.js"; -import { syncDocumentMetadata, countDocumentsPendingMetadata } from "../metadata-store.js"; +import { + syncDocumentMetadata, + countDocumentsPendingMetadata, + listMetadata, + MetadataBindingBudgetError, + type ListMetadataOptions, + type ListMetadataResult, +} from "../metadata-store.js"; +import { formatMetadataKeySummaries } from "../metadata-format.js"; import type { DocumentMetadata } from "../metadata.js"; -import { parseMetadataFilter, type MetadataFilter } from "../metadata-filter.js"; +import { parseMetadataFilter, parseMetadataMatch, type MetadataFilter, type MetadataMatch } from "../metadata-filter.js"; import { disposeDefaultLlamaCpp, getDefaultLlamaCpp, setDefaultLlamaCpp, LlamaCpp, withLLMSession, pullModels, DEFAULT_MODEL_CACHE_DIR, resolveEmbedModel, resolveGenerateModel, resolveRerankModel, resolveModels, inspectGgufFile, isDarwinMetalMitigationActive } from "../llm.js"; import { formatSearchResults, @@ -1916,6 +1924,128 @@ function collectionRename(oldName: string, newName: string): void { console.log(` Virtual paths updated: ${c.cyan}qmd://${oldName}/${c.reset} → ${c.cyan}qmd://${newName}/${c.reset}`); } +// Metadata discovery drill-down. Collection names are already validated; +// an empty list means the default collections resolved to nothing, which +// the store reads as "every collection" exactly as search does. +function collectionMetadata(collectionNames: string[], options: ListMetadataOptions): void { + const db = getDb(); + + if (listCollections(db).length === 0) { + console.log("No collections found. Run 'qmd collection add .' to create one."); + closeDb(); + return; + } + + // Discovery always applies the extraction gate, filter or not. + warnPendingMetadata(db); + + let result: ListMetadataResult; + try { + result = listMetadata(db, { ...options, collection: collectionSearchFilter(collectionNames) }); + } catch (error) { + if (!(error instanceof MetadataBindingBudgetError)) throw error; + closeDb(); + console.error(`${c.yellow}${error.message}${c.reset}`); + process.exit(1); + } + closeDb(); + + const selection = options.match || options.filter; + console.log(formatMetadataKeySummaries(result, { + showCollections: collectionNames.length !== 1, + valueWindowHint: "--value-limit , --value-offset , or --all-values", + keyWindowHint: "--key-limit , --key-offset , or --all-keys", + keyOffset: options.keyOffset ?? 0, + keyOffsetLabel: "--key-offset", + emptyMessage: selection + ? "No metadata matches. Run 'qmd collection metadata' without --match or --filter to see every key." + : "No metadata found. Add qmd.metadata frontmatter and run 'qmd update'.", + colors: c, + })); +} + +// Parse the discovery-specific flags; exits with usage on a bad value. +function parseCliMetadataOptions(values: Record): ListMetadataOptions { + const options: ListMetadataOptions = { + match: parseCliMetadataMatch(values.match), + filter: parseCliMetadataFilter(values.filter), + }; + + // Discovery windows two dimensions, so the single-window search flags + // have no reading here. Point at the flags that do. + if (values.n !== undefined) { + console.error("-n is not an option of 'qmd collection metadata'"); + console.error("Use --value-limit for values per key, or --key-limit for keys"); + process.exit(1); + } + if (values.all) { + console.error("--all is not an option of 'qmd collection metadata'"); + console.error("Use --all-values, --all-keys, or both"); + process.exit(1); + } + + const formatAlias = ["json", "csv", "md", "xml", "files"].find(flag => values[flag]); + const format = typeof values.format === "string" ? values.format.trim().toLowerCase() : undefined; + if (formatAlias || (format !== undefined && format !== "cli")) { + console.error(`${formatAlias ? `--${formatAlias}` : `--format ${String(values.format)}`} is not supported by 'qmd collection metadata'`); + console.error("This command prints text. Use the SDK, MCP metadata tool, or POST /metadata for structured output"); + process.exit(1); + } + + if (values["all-keys"]) { + options.keyLimit = Infinity; + } else if (values["key-limit"] !== undefined) { + options.keyLimit = parsePositiveInteger(values["key-limit"], "--key-limit"); + } + if (values["key-offset"] !== undefined) { + options.keyOffset = parseNonNegativeInteger(values["key-offset"], "--key-offset"); + } + + if (values["all-values"]) { + options.valueLimit = Infinity; + } else if (values["value-limit"] !== undefined) { + options.valueLimit = parsePositiveInteger(values["value-limit"], "--value-limit"); + } + if (values["value-offset"] !== undefined) { + options.valueOffset = parseNonNegativeInteger(values["value-offset"], "--value-offset"); + } + + if (values["min-count"] !== undefined) { + options.minCount = parsePositiveInteger(values["min-count"], "--min-count"); + } + + if (values.sort !== undefined) { + if (values.sort !== "count" && values.sort !== "value") { + console.error(`Invalid --sort value: ${String(values.sort)}`); + console.error("Valid: count, value"); + process.exit(1); + } + options.sort = values.sort; + } + + return options; +} + +function parsePositiveInteger(raw: unknown, flag: string): number { + const parsed = Number(raw); + if (!Number.isSafeInteger(parsed) || parsed < 1) { + console.error(`Invalid ${flag} value: ${String(raw)}`); + console.error(`${flag} must be a positive safe integer`); + process.exit(1); + } + return parsed; +} + +function parseNonNegativeInteger(raw: unknown, flag: string): number { + const parsed = Number(raw); + if (!Number.isSafeInteger(parsed) || parsed < 0) { + console.error(`Invalid ${flag} value: ${String(raw)}`); + console.error(`${flag} must be a non-negative safe integer`); + process.exit(1); + } + return parsed; +} + async function indexFiles(pwd?: string, globPattern: string = DEFAULT_GLOB, collectionName?: string, suppressEmbedNotice: boolean = false, ignorePatterns?: string[]): Promise { const db = getDb(); const resolvedPwd = pwd || getPwd(); @@ -2825,19 +2955,43 @@ function parseStructuredQuery(query: string): ParsedStructuredQuery | null { // Parse and validate a --filter JSON string; exits with an actionable // message on malformed JSON or an invalid filter AST. function parseCliMetadataFilter(rawFilter: unknown): MetadataFilter | undefined { - if (rawFilter === undefined) return undefined; + return parseCliPredicateFlag(rawFilter, { + flag: "--filter", + example: `{"field":"status","operator":"eq","value":"published"}`, + parse: parseMetadataFilter, + }); +} + +// Same grammar as --filter, evaluated against metadata entries for discovery. +function parseCliMetadataMatch(rawMatch: unknown): MetadataMatch | undefined { + return parseCliPredicateFlag(rawMatch, { + flag: "--match", + example: `{"field":"key","operator":"eq","value":"topics"}`, + parse: parseMetadataMatch, + }); +} + +interface CliPredicateFlag { + flag: string; + example: string; + parse: (input: unknown) => Predicate; +} - let filterJson: unknown; +// Parse a JSON predicate flag with the parser for its record type. +function parseCliPredicateFlag(raw: unknown, predicateFlag: CliPredicateFlag): Predicate | undefined { + if (raw === undefined) return undefined; + + let astJson: unknown; try { - filterJson = JSON.parse(String(rawFilter)); + astJson = JSON.parse(String(raw)); } catch (err) { - console.error(`Invalid --filter JSON: ${err instanceof Error ? err.message : String(err)}`); - console.error(`Example: --filter '{"field":"status","operator":"eq","value":"published"}'`); + console.error(`Invalid ${predicateFlag.flag} JSON: ${err instanceof Error ? err.message : String(err)}`); + console.error(`Example: ${predicateFlag.flag} '${predicateFlag.example}'`); process.exit(1); } try { - return parseMetadataFilter(filterJson); + return predicateFlag.parse(astJson); } catch (err) { console.error(err instanceof Error ? err.message : String(err)); process.exit(1); @@ -3112,7 +3266,17 @@ function parseCLI() { json: { type: "boolean" }, explain: { type: "boolean" }, collection: { type: "string", short: "c", multiple: true }, // Filter by collection(s) - filter: { type: "string" }, // Metadata filter (JSON AST) for search/vsearch/query + filter: { type: "string" }, // Metadata filter (JSON AST) for search/vsearch/query/collection metadata + // Metadata discovery options (collection metadata) + match: { type: "string" }, // Metadata match (JSON AST) over the entries reported + "key-limit": { type: "string" }, // keys reported (default 50) + "key-offset": { type: "string" }, // keys skipped before the window + "all-keys": { type: "boolean" }, // remove the key window + "value-limit": { type: "string" }, // values reported per key (default 10) + "value-offset": { type: "string" }, // values skipped per key before the window + "all-values": { type: "boolean" }, // remove the value window + sort: { type: "string" }, // count (default) | value + "min-count": { type: "string" }, // drop values held by fewer documents // Collection options name: { type: "string" }, // collection name mask: { type: "string" }, // glob pattern @@ -3652,6 +3816,7 @@ function showHelp(): void { console.log(""); console.log("Collections & context:"); console.log(" qmd collection add/list/remove/rename/show - Manage indexed folders"); + console.log(" qmd collection metadata [name] [--match J] - Discover metadata keys and values to filter on"); console.log(" qmd context add/list/rm - Attach human-written summaries"); console.log(" qmd ls [collection[/path]] - Inspect indexed files"); console.log(""); @@ -4647,6 +4812,15 @@ if (isMain) { break; } + case "metadata": { + // Positional names are optional; omitted means the default + // collections, as an unscoped search does. + const rawNames = cli.args.length > 1 ? cli.args.slice(1) : undefined; + const collectionNames = resolveCollectionFilter(rawNames, true); + collectionMetadata(collectionNames, parseCliMetadataOptions(cli.values)); + break; + } + case "help": case undefined: { console.log("Usage: qmd collection [options]"); @@ -4657,6 +4831,13 @@ if (isMain) { console.log(" remove Remove a collection"); console.log(" rename Rename a collection"); console.log(" show Show collection details"); + console.log(" metadata [name...] Discover metadata keys, types, and value counts"); + console.log(" --match Report only metadata entries matching this condition"); + console.log(" (same AST as --filter; 'field' is the entry's key or value)"); + console.log(" --filter Count only documents matching a metadata filter"); + console.log(" --key-limit Keys reported (default 50), --key-offset pages, --all-keys removes the window"); + console.log(" --value-limit Values per key (default 10), --value-offset pages, --all-values removes the window"); + console.log(" --sort count|value Value order (default count), --min-count drops the tail"); console.log(" update-cmd [cmd] Set pre-update command (e.g., 'git pull')"); console.log(" include Include in default queries"); console.log(" exclude Exclude from default queries"); @@ -4666,6 +4847,9 @@ if (isMain) { console.log(" qmd collection add ~/notes --name notes --mask 'a.md,journals/*.md'"); console.log(" qmd collection update-cmd brain 'git pull'"); console.log(" qmd collection exclude archive"); + console.log(" qmd collection metadata notes --match '{\"field\":\"key\",\"operator\":\"eq\",\"value\":\"topics\"}'"); + console.log(" qmd collection metadata notes --match '{\"field\":\"value\",\"operator\":\"eq\",\"value\":\"docs-team\"}'"); + console.log(" qmd collection metadata notes --match '{\"field\":\"key\",\"operator\":\"eq\",\"value\":\"topics\"}' --filter '{\"field\":\"status\",\"operator\":\"eq\",\"value\":\"published\"}'"); process.exit(0); } diff --git a/src/metadata-format.ts b/src/metadata-format.ts new file mode 100644 index 000000000..1d2ba8cc1 --- /dev/null +++ b/src/metadata-format.ts @@ -0,0 +1,226 @@ +/** + * QMD Metadata Format - Plain-text rendering of metadata discovery results. + * + * Shared by the CLI and the MCP `metadata` tool so both print one shape: a + * filter line when a filter narrowed the documents, a header per key, a body + * per type, and footers naming the remainder and the options that reach it + * whenever a value list or the key list is windowed. The empty states render + * here too, so the filter line survives them. + * + * String values print bare only when the bare form is unambiguous. A value + * that is empty, padded, contains a quote, a backslash, a control character, + * a line separator, or a delimiter of the compact `value (count), ...` list, + * or that reads as a JSON number, boolean, or null, or as the list's + * remainder tail, is printed as a JSON string, so what the reader sees is + * exactly the value. + */ + +import type { + ListMetadataResult, + MetadataKeySummary, + MetadataKeyTypeSummary, + MetadataValueCount, +} from "./metadata-store.js"; + +export interface MetadataFormatColors { + reset: string; + dim: string; + bold: string; + cyan: string; +} + +export interface FormatMetadataOptions { + /** Show which collections contribute each type on type-split keys. */ + showCollections?: boolean; + /** How the caller widens or pages the value window, e.g. `--value-limit , --value-offset , or --all-values`. */ + valueWindowHint: string; + /** How the caller widens or pages the key window, e.g. `--key-limit , --key-offset , or --all-keys`. */ + keyWindowHint: string; + /** The key offset the caller requested, and how the caller spells that option, e.g. `--key-offset`. */ + keyOffset: number; + keyOffsetLabel: string; + /** Printed in place of the key blocks when no metadata entry is in the result at all. */ + emptyMessage: string; + /** ANSI sequences; omit for plain text. */ + colors?: MetadataFormatColors; +} + +const NO_COLORS: MetadataFormatColors = { reset: "", dim: "", bold: "", cyan: "" }; + +/** + * Render the whole result: the filter line when a filter is in play, then + * one block per key separated by blank lines and the key footer when keys + * were left out of the window, or the one-line empty state when there are no + * blocks to show. + */ +export function formatMetadataKeySummaries(result: ListMetadataResult, options: FormatMetadataOptions): string { + const colors = options.colors ?? NO_COLORS; + const blocks: string[] = []; + + if (result.filteredDocuments !== undefined) { + blocks.push(`${colors.dim}filter:${colors.reset} ${formatCount(result.filteredDocuments)} of ${formatCount(result.documents)} documents`); + } + + if (result.totalKeys === 0) { + blocks.push(`${colors.dim}${options.emptyMessage}${colors.reset}`); + } else if (result.keys.length === 0) { + const keyLabel = result.totalKeys === 1 ? "key" : "keys"; + blocks.push(`${colors.dim}No keys at ${options.keyOffsetLabel} ${formatCount(options.keyOffset)}, ${formatCount(result.totalKeys)} ${keyLabel} in total.${colors.reset}`); + } + + blocks.push(...result.keys.map(summary => formatMetadataKeySummary(summary, result, options))); + + if (result.remainingKeys > 0) { + blocks.push(`${colors.dim}${formatCount(result.remainingKeys)} more ${result.remainingKeys === 1 ? "key" : "keys"}, use ${options.keyWindowHint}${colors.reset}`); + } + + return blocks.join("\n\n"); +} + +/** + * One key: a header of name, types, coverage, and distinct count, then the + * body per type. Coverage is measured against the documents the filter + * admitted when there is one, otherwise against every active document. + */ +export function formatMetadataKeySummary(summary: MetadataKeySummary, result: ListMetadataResult, options: FormatMetadataOptions): string { + const colors = options.colors ?? NO_COLORS; + const typeLabel = summary.types.map(typeLabelOf).join(" | "); + const coverage = `${formatCount(summary.documents)} of ${formatCount(result.filteredDocuments ?? result.documents)} documents`; + const header = [`${colors.cyan}${colors.bold}${summary.key}${colors.reset}`, `${colors.dim}${typeLabel}${colors.reset}`, coverage]; + const lines: string[] = []; + + if (summary.types.length === 1) { + const typeSummary = summary.types[0]!; + if (typeSummary.type !== "boolean") header.push(`${formatCount(typeSummary.distinctValues)} distinct`); + lines.push(header.join(" "), ...formatTypeBody(typeSummary)); + } else { + lines.push(header.join(" "), ...formatTypeSplit(summary.types, options)); + } + + const remainingValues = summary.types.reduce((sum, typeSummary) => sum + typeSummary.remainingValues, 0); + if (remainingValues > 0) { + lines.push(`${colors.dim}${formatCount(remainingValues)} more ${remainingValues === 1 ? "value" : "values"}, use ${options.valueWindowHint}${colors.reset}`); + } + + return lines.join("\n"); +} + +/** Body for a key with one type: vertical values for strings, a range for numbers, one line for booleans. */ +function formatTypeBody(typeSummary: MetadataKeyTypeSummary): string[] { + if (typeSummary.type === "boolean") return [` ${formatBooleanCounts(typeSummary.values)}`]; + + if (typeSummary.type === "number") { + const lines = [` ${formatRange(typeSummary)}`]; + if (typeSummary.values.length > 0) lines.push(` ${formatInlineValues(typeSummary).join(" ")}`); + return lines; + } + + const valueWidth = Math.max(...typeSummary.values.map(count => formatValue(count.value).length)); + const countWidth = Math.max(...typeSummary.values.map(count => formatCount(count.documents).length)); + return typeSummary.values.map(count => ` ${formatValue(count.value).padEnd(valueWidth)} ${formatCount(count.documents).padStart(countWidth)}`); +} + +/** + * Body for a key whose documents disagree on type: one line per type with + * its own document count and a compact value summary, plus the contributing + * collections when the view spans more than one. + */ +function formatTypeSplit(typeSummaries: MetadataKeyTypeSummary[], options: FormatMetadataOptions): string[] { + const typeWidth = Math.max(...typeSummaries.map(typeSummary => typeSummary.type.length)); + const documentsWidth = Math.max(...typeSummaries.map(typeSummary => formatCount(typeSummary.documents).length)); + + const rows = typeSummaries.map(typeSummary => { + let valuesSummary: string; + if (typeSummary.type === "boolean") valuesSummary = formatBooleanCounts(typeSummary.values); + else if (typeSummary.type === "number") valuesSummary = formatRange(typeSummary); + else { + const inlineValues = typeSummary.values.map(count => `${formatValue(count.value)} (${formatCount(count.documents)})`); + if (typeSummary.remainingValues > 0) inlineValues.push(`${formatCount(typeSummary.remainingValues)} more`); + valuesSummary = inlineValues.join(", "); + } + const documentsLabel = typeSummary.documents === 1 ? "doc " : "docs"; + const row = ` ${typeSummary.type.padEnd(typeWidth)} ${formatCount(typeSummary.documents).padStart(documentsWidth)} ${documentsLabel} ${valuesSummary}`; + return { row, collections: typeSummary.collections.join(", ") }; + }); + + if (!options.showCollections) return rows.map(({ row }) => row); + + const rowWidth = Math.max(...rows.map(({ row }) => row.length)); + return rows.map(({ row, collections }) => `${row.padEnd(rowWidth)} ${collections}`); +} + +function typeLabelOf(typeSummary: MetadataKeyTypeSummary): string { + return typeSummary.multiValued ? `${typeSummary.type}[]` : typeSummary.type; +} + +function formatRange(typeSummary: MetadataKeyTypeSummary): string { + const range = typeSummary.range!; + return `min ${formatValue(range.min)} median ${formatValue(range.median)} max ${formatValue(range.max)}`; +} + +/** + * Numbers enumerate inline as `value (documents)`. When the whole + * distribution fits it reads in value order; a truncated window keeps the + * requested order so the most common values stay visible. + */ +function formatInlineValues(typeSummary: MetadataKeyTypeSummary): string[] { + const values = typeSummary.remainingValues === 0 + ? [...typeSummary.values].sort((a, b) => Number(a.value) - Number(b.value)) + : typeSummary.values; + return values.map(count => `${formatValue(count.value)} (${formatCount(count.documents)})`); +} + +function formatBooleanCounts(values: MetadataValueCount[]): string { + const trueCount = values.find(count => count.value === true); + const falseCount = values.find(count => count.value === false); + const parts: string[] = []; + if (trueCount) parts.push(`true ${formatCount(trueCount.documents)}`); + if (falseCount) parts.push(`false ${formatCount(falseCount.documents)}`); + return parts.join(" "); +} + +function formatValue(value: string | number | boolean): string { + if (typeof value !== "string") return String(value); + return isUnambiguousBare(value) ? value : formatQuoted(value); +} + +// Controls (C0, DEL, C1) and the line and paragraph separators, none of which +// print as themselves. +const CONTROL_CHARACTERS = /[\u0000-\u001f\u007f-\u009f\u2028\u2029]/u; +// Whitespace other than one interior ASCII space between other characters. +const AMBIGUOUS_WHITESPACE = /^\s|\s$|\s\s|[^\S ]/u; +// The compact list joins `value (count)` items with ", " and ends with a +// remainder tail, so a bare string can neither contain the delimiters nor +// read as a tail. One rule for both layouts, so a value never prints two ways. +const LIST_DELIMITERS = /[,()]/u; +const LIST_TAIL = /^(?:\.\.\.|\d+ more)$/u; + +/** True when printing the string as-is cannot be mistaken for another value or for layout. */ +function isUnambiguousBare(text: string): boolean { + if (text === "" || text.includes('"') || text.includes("\\")) return false; + if (CONTROL_CHARACTERS.test(text) || AMBIGUOUS_WHITESPACE.test(text)) return false; + if (LIST_DELIMITERS.test(text) || LIST_TAIL.test(text)) return false; + return !readsAsJsonLiteral(text); +} + +/** A string that JSON would parse as something other than a string, e.g. `42`, `true`, `null`, `[1]`. */ +function readsAsJsonLiteral(text: string): boolean { + try { + JSON.parse(text); + return true; + } catch { + return false; + } +} + +/** A JSON string, with every character in CONTROL_CHARACTERS escaped, not only the ones JSON.stringify escapes. */ +function formatQuoted(text: string): string { + return JSON.stringify(text).replace( + /[\u007f-\u009f\u2028\u2029]/gu, + character => `\\u${character.codePointAt(0)!.toString(16).padStart(4, "0")}`, + ); +} + +function formatCount(count: number): string { + return count.toLocaleString("en-US"); +} diff --git a/test/metadata-cli.test.ts b/test/metadata-cli.test.ts index 5334fed34..2741aa1c2 100644 --- a/test/metadata-cli.test.ts +++ b/test/metadata-cli.test.ts @@ -159,3 +159,334 @@ describe("qmd search --filter", () => { expect(stderr).toMatch(/unknown operator 'equal'/); }, 30000); }); + +describe("qmd collection metadata", () => { + beforeAll(async () => { + await writeFile(join(fixturesDir, "discovery.md"), [ + "---", + "qmd:", + " metadata:", + " status: published", + " topics: [typescript, sqlite, search, architecture, embeddings, mcp, cli, testing, agents, indexing, chunking, reranking]", + " priority: 5", + " owner: docs-team", + "---", + "", + "# Discovery doc", + "", + "cli filter keyword body", + "", + ].join("\n")); + await writeFile(join(fixturesDir, "priority.md"), [ + "---", + "qmd:", + " metadata:", + " status: draft", + " priority: 1", + " reviewed: false", + " reviewers: [docs-team, security-team]", + "---", + "", + "# Priority doc", + "", + "cli filter keyword body", + "", + ].join("\n")); + // Nothing enforces a type across documents, so one file in the same + // collection may spell priority as a label where the others use numbers. + await writeFile(join(fixturesDir, "conflict.md"), [ + "---", + "qmd:", + " metadata:", + " status: archived", + " priority: high", + "---", + "", + "# Conflict doc", + "", + "cli filter keyword body", + "", + ].join("\n")); + const updateResult = await runQmd(["update"]); + expect(updateResult.exitCode).toBe(0); + }, 60000); + + test("lists every key with coverage, type, and a value window", async () => { + const { stdout, exitCode } = await runQmd(["collection", "metadata", "notes"]); + expect(exitCode).toBe(0); + + // Keys arrive in coverage order; the denominator is every active document. + expect(stdout).toMatch(/^status {2}string {2}5 of 6 documents {2}3 distinct\n {2}draft {6}2\n {2}published {2}2\n {2}archived {3}1\n/); + expect(stdout).toContain("topics string[] 2 of 6 documents 13 distinct"); + expect(stdout).toContain("reviewed boolean 1 of 6 documents\n false 1"); + expect(stdout).toContain("reviewers string[] 1 of 6 documents 2 distinct\n docs-team 1\n security-team 1"); + }, 30000); + + test("splits a key whose documents disagree on type, even within one collection", async () => { + const { stdout, exitCode } = await runQmd(["collection", "metadata", "notes", "--match", '{"field":"key","operator":"eq","value":"priority"}']); + expect(exitCode).toBe(0); + + // One row per type with its own document count. The collection column + // is omitted since the view covers a single collection. + expect(stdout).toBe([ + "priority number | string 3 of 6 documents", + " number 2 docs min 1 median 3 max 5", + " string 1 doc high (1)", + "", + ].join("\n")); + }, 30000); + + test("truncates value lists with an explicit remainder and escape hatch", async () => { + const { stdout, exitCode } = await runQmd(["collection", "metadata", "notes", "--match", '{"field":"key","operator":"eq","value":"topics"}']); + expect(exitCode).toBe(0); + + const valueLines = stdout.split("\n").filter(line => /^ {2}\S/.test(line)); + expect(valueLines).toHaveLength(10); + expect(valueLines[0]).toBe(" typescript 2"); + expect(stdout).toContain("3 more values, use --value-limit , --value-offset , or --all-values"); + expect(stdout).not.toContain("status"); + }, 30000); + + test("--value-limit, --value-offset, and --all-values size and page the value window", async () => { + const topics = ["--match", '{"field":"key","operator":"eq","value":"topics"}']; + + const limited = await runQmd(["collection", "metadata", "notes", ...topics, "--value-limit", "2"]); + expect(limited.stdout).toContain("11 more values, use --value-limit , --value-offset , or --all-values"); + + const paged = await runQmd(["collection", "metadata", "notes", ...topics, "--value-limit", "2", "--value-offset", "11"]); + expect(paged.stdout.split("\n").filter(line => /^ {2}\S/.test(line))).toHaveLength(2); + expect(paged.stdout).not.toContain("more values"); + + const all = await runQmd(["collection", "metadata", "notes", ...topics, "--all-values"]); + expect(all.stdout).not.toContain("more values"); + expect(all.stdout.split("\n").filter(line => /^ {2}\S/.test(line))).toHaveLength(13); + }, 30000); + + test("--key-limit, --key-offset, and --all-keys size and page the key window", async () => { + const keyHeaders = (stdout: string) => stdout.split("\n").filter(line => /^\S.* of 6 documents/.test(line)).map(line => line.split(" ")[0]); + + const everything = await runQmd(["collection", "metadata", "notes"]); + expect(keyHeaders(everything.stdout)).toHaveLength(6); + expect(everything.stdout).not.toContain("more keys"); + + const limited = await runQmd(["collection", "metadata", "notes", "--key-limit", "2"]); + expect(keyHeaders(limited.stdout)).toEqual(["status", "priority"]); + expect(limited.stdout).toContain("4 more keys, use --key-limit , --key-offset , or --all-keys"); + + const paged = await runQmd(["collection", "metadata", "notes", "--key-limit", "2", "--key-offset", "2"]); + expect(keyHeaders(paged.stdout)).toEqual(["topics", "owner"]); + expect(paged.stdout).toContain("2 more keys, use --key-limit , --key-offset , or --all-keys"); + + const lastPage = await runQmd(["collection", "metadata", "notes", "--key-limit", "2", "--key-offset", "4"]); + expect(keyHeaders(lastPage.stdout)).toEqual(["reviewed", "reviewers"]); + expect(lastPage.stdout).not.toContain("more keys"); + + const pastTheEnd = await runQmd(["collection", "metadata", "notes", "--key-offset", "6"]); + expect(pastTheEnd.exitCode).toBe(0); + expect(pastTheEnd.stdout).toContain("No keys at --key-offset 6, 6 keys in total."); + + const all = await runQmd(["collection", "metadata", "notes", "--all-keys", "--key-limit", "1"]); + expect(keyHeaders(all.stdout)).toHaveLength(6); + }, 60000); + + test("a match on the value field is a reverse lookup across keys", async () => { + const { stdout, exitCode } = await runQmd(["collection", "metadata", "notes", "--match", '{"field":"value","operator":"eq","value":"docs-team"}']); + expect(exitCode).toBe(0); + + expect(stdout).toBe([ + "owner string 1 of 6 documents 1 distinct", + " docs-team 1", + "", + "reviewers string[] 1 of 6 documents 1 distinct", + " docs-team 1", + "", + ].join("\n")); + }, 30000); + + test("--filter counts only matching documents and says so in the header", async () => { + const { stdout, exitCode } = await runQmd([ + "collection", "metadata", "notes", "--match", '{"field":"key","operator":"eq","value":"priority"}', + "--filter", '{"field":"status","operator":"eq","value":"published"}', + ]); + expect(exitCode).toBe(0); + // The filter line reports the population, and every key's coverage is + // measured against it. Two documents are published and one of them has + // a priority, so the type split above collapses to a flat number view. + expect(stdout).toBe([ + "filter: 2 of 6 documents", + "", + "priority number 1 of 2 documents 1 distinct", + " min 5 median 5 max 5", + " 5 (1)", + "", + ].join("\n")); + }, 30000); + + test("--sort value and --min-count reshape the window", async () => { + const sorted = await runQmd(["collection", "metadata", "notes", "--match", '{"field":"key","operator":"eq","value":"topics"}', "--sort", "value", "--value-limit", "2"]); + expect(sorted.stdout).toContain(" agents 1\n architecture 1\n"); + + const common = await runQmd(["collection", "metadata", "notes", "--match", '{"field":"key","operator":"eq","value":"topics"}', "--min-count", "2"]); + expect(common.stdout).toContain("topics string[] 2 of 6 documents 1 distinct\n typescript 2\n"); + expect(common.stdout).not.toContain("more values"); + }, 30000); + + test("omitting the collection covers the default collections", async () => { + const { stdout, exitCode } = await runQmd(["collection", "metadata"]); + expect(exitCode).toBe(0); + expect(stdout).toContain("status string 5 of 6 documents 3 distinct"); + }, 30000); + + test("a composed match selects entries across the key and value fields", async () => { + const { stdout, exitCode } = await runQmd([ + "collection", "metadata", "notes", + "--match", JSON.stringify({ + operator: "and", + operands: [ + { field: "key", operator: "eq", value: "topics" }, + { operator: "or", operands: [ + { field: "value", operator: "prefix", value: "sql" }, + { field: "value", operator: "contains", value: "ARCH", caseInsensitive: true }, + ] }, + ], + }), + ]); + expect(exitCode).toBe(0); + expect(stdout).toBe([ + "topics string[] 1 of 6 documents 3 distinct", + " architecture 1", + " search 1", + " sqlite 1", + "", + ].join("\n")); + }, 30000); + + test("a type match reports one side of a split key", async () => { + const { stdout, exitCode } = await runQmd([ + "collection", "metadata", "notes", + "--match", '{"field":"value","operator":"type","value":"number"}', + ]); + expect(exitCode).toBe(0); + expect(stdout).toBe([ + "priority number 2 of 6 documents 2 distinct", + " min 1 median 3 max 5", + " 1 (1) 5 (1)", + "", + ].join("\n")); + }, 30000); + + test("reports when nothing matches", async () => { + const { stdout, exitCode } = await runQmd([ + "collection", "metadata", "notes", "--match", '{"field":"key","operator":"prefix","value":"missing-"}', + ]); + expect(exitCode).toBe(0); + expect(stdout).toContain("No metadata matches"); + }, 30000); + + test("keeps the filter line when the filtered documents declare no metadata", async () => { + // plain.md is the one document without a status, and it has no metadata at all. + const { stdout, exitCode } = await runQmd([ + "collection", "metadata", "notes", "--filter", '{"field":"status","operator":"exists","value":false}', + ]); + expect(exitCode).toBe(0); + expect(stdout).toBe([ + "filter: 1 of 6 documents", + "", + "No metadata matches. Run 'qmd collection metadata' without --match or --filter to see every key.", + "", + ].join("\n")); + }, 30000); + + test("exits on an unknown collection", async () => { + const { stderr, exitCode } = await runQmd(["collection", "metadata", "missing"]); + expect(exitCode).toBe(1); + expect(stderr).toContain("Collection not found: missing"); + }, 30000); + + test("exits on an invalid filter", async () => { + const { stderr, exitCode } = await runQmd([ + "collection", "metadata", "notes", "--filter", '{"field":"status","operator":"equal","value":"x"}', + ]); + expect(exitCode).toBe(1); + expect(stderr).toMatch(/Invalid metadata filter at \$/); + }, 30000); + + test("exits on an invalid match, naming the match", async () => { + const badJson = await runQmd(["collection", "metadata", "notes", "--match", "{nope"]); + expect(badJson.exitCode).toBe(1); + expect(badJson.stderr).toMatch(/Invalid --match JSON/); + expect(badJson.stderr).toContain(`Example: --match '{"field":"key","operator":"eq","value":"topics"}'`); + + const badField = await runQmd([ + "collection", "metadata", "notes", "--match", '{"field":"topics","operator":"eq","value":"x"}', + ]); + expect(badField.exitCode).toBe(1); + expect(badField.stderr).toMatch(/Invalid metadata match at \$: 'topics' is not a field of a metadata entry, expected 'key' or 'value'/); + + const badOperator = await runQmd([ + "collection", "metadata", "notes", "--match", '{"field":"value","operator":"exists","value":true}', + ]); + expect(badOperator.exitCode).toBe(1); + expect(badOperator.stderr).toMatch(/'exists' has no meaning for a single metadata entry/); + }, 30000); + + test("exits on invalid window, --min-count, and --sort values", async () => { + const badLimit = await runQmd(["collection", "metadata", "notes", "--value-limit", "0"]); + expect(badLimit.exitCode).toBe(1); + expect(badLimit.stderr).toContain("Invalid --value-limit value: 0"); + + const badKeyLimit = await runQmd(["collection", "metadata", "notes", "--key-limit", "x"]); + expect(badKeyLimit.exitCode).toBe(1); + expect(badKeyLimit.stderr).toContain("Invalid --key-limit value: x"); + + const badOffset = await runQmd(["collection", "metadata", "notes", "--key-offset=-1"]); + expect(badOffset.exitCode).toBe(1); + expect(badOffset.stderr).toContain("Invalid --key-offset value: -1"); + expect(badOffset.stderr).toContain("--key-offset must be a non-negative safe integer"); + + const badMinCount = await runQmd(["collection", "metadata", "notes", "--min-count", "x"]); + expect(badMinCount.exitCode).toBe(1); + expect(badMinCount.stderr).toContain("Invalid --min-count value: x"); + + const badSort = await runQmd(["collection", "metadata", "notes", "--sort", "size"]); + expect(badSort.exitCode).toBe(1); + expect(badSort.stderr).toContain("Invalid --sort value: size"); + }, 30000); + + test("rejects unsafe integers with usage, preserving the input rather than its rounded value", async () => { + for (const flag of ["--key-limit", "--key-offset", "--value-limit", "--value-offset", "--min-count"]) { + const result = await runQmd(["collection", "metadata", "notes", flag, "9007199254740993"]); + expect(result.exitCode).toBe(1); + expect(result.stdout).toBe(""); + expect(result.stderr).toContain(`Invalid ${flag} value: 9007199254740993`); + expect(result.stderr).toContain("safe integer"); + expect(result.stderr).not.toContain("MetadataOptionError"); + expect(result.stderr).not.toContain("at collectionMetadata"); + } + }, 60000); + + test("rejects unsupported formats rather than silently returning text", async () => { + for (const flags of [["--json"], ["--format", "json"], ["--csv"], ["--md"], ["--xml"], ["--files"]]) { + const result = await runQmd(["collection", "metadata", "notes", ...flags]); + expect(result.exitCode).toBe(1); + expect(result.stdout).toBe(""); + expect(result.stderr).toContain("is not supported by 'qmd collection metadata'"); + expect(result.stderr).toContain("Use the SDK, MCP metadata tool, or POST /metadata for structured output"); + } + const result = await runQmd(["collection", "metadata", "notes", "--format", "cli"]); + expect(result.exitCode).toBe(0); + expect(result.stdout).toContain("topics string[]"); + }, 60000); + + test("points the search window flags at the discovery ones", async () => { + const n = await runQmd(["collection", "metadata", "notes", "-n", "3"]); + expect(n.exitCode).toBe(1); + expect(n.stderr).toContain("-n is not an option of 'qmd collection metadata'"); + expect(n.stderr).toContain("Use --value-limit for values per key, or --key-limit for keys"); + + const all = await runQmd(["collection", "metadata", "notes", "--all"]); + expect(all.exitCode).toBe(1); + expect(all.stderr).toContain("--all is not an option of 'qmd collection metadata'"); + expect(all.stderr).toContain("Use --all-values, --all-keys, or both"); + }, 30000); +}); diff --git a/test/metadata-format.test.ts b/test/metadata-format.test.ts new file mode 100644 index 000000000..8cde1ee9b --- /dev/null +++ b/test/metadata-format.test.ts @@ -0,0 +1,139 @@ +/** + * Metadata discovery rendering: the plain-text shape the CLI and the MCP + * `metadata` tool share, exercised directly where the CLI fixture cannot + * reach (string identity, control characters, empty states under a filter). + */ + +import { describe, test, expect } from "vitest"; +import { formatMetadataKeySummaries, type FormatMetadataOptions } from "../src/metadata-format.js"; +import type { ListMetadataResult, MetadataKeyTypeSummary, MetadataScalar } from "../src/metadata-store.js"; + +const OPTIONS: FormatMetadataOptions = { + valueWindowHint: "--value-limit ", + keyWindowHint: "--key-limit ", + keyOffset: 0, + keyOffsetLabel: "--key-offset", + emptyMessage: "No metadata matches.", +}; + +function stringTypeSummary(values: string[], remainingValues = 0): MetadataKeyTypeSummary { + return { + type: "string", + multiValued: false, + documents: values.length, + distinctValues: values.length + remainingValues, + values: values.map(value => ({ value, documents: 1 })), + remainingValues, + collections: ["notes"], + }; +} + +function stringResult(values: string[]): ListMetadataResult { + return { documents: values.length, totalKeys: 1, keys: [{ key: "label", documents: values.length, types: [stringTypeSummary(values)] }], remainingKeys: 0 }; +} + +/** A key held as a string by some documents and a number by one, which prints the compact `value (count)` list. */ +function splitResult(values: string[], remainingValues = 0): ListMetadataResult { + const numberSummary: MetadataKeyTypeSummary = { + type: "number", multiValued: false, documents: 1, distinctValues: 1, + values: [{ value: 7, documents: 1 }], remainingValues: 0, range: { min: 7, median: 7, max: 7 }, collections: ["notes"], + }; + const documents = values.length + 1; + return { documents, totalKeys: 1, keys: [{ key: "label", documents, types: [stringTypeSummary(values, remainingValues), numberSummary] }], remainingKeys: 0 }; +} + +/** + * The items of the compact list on the string row of a type split, read back + * the way a reader must: a quoted item is a JSON string (so it may contain the + * delimiters), a bare item runs to its count, and the tail closes the list. + */ +function compactItems(output: string): { value: MetadataScalar; documents: number }[] { + const row = output.split("\n").find(line => line.startsWith(" string "))!; + const list = row.replace(/^ {2}string {2}\d+ docs? {2}/u, ""); + const items: { value: MetadataScalar; documents: number }[] = []; + let rest = list; + while (rest !== "") { + const tail = /^(\d+) more$/u.exec(rest); + if (tail) { + items.push({ value: `${tail[1]} more`, documents: -1 }); + break; + } + const item = rest.startsWith('"') + ? /^("(?:[^"\\]|\\.)*") \((\d+)\)(?:, |$)/u.exec(rest)! + : /^([^,()]*) \((\d+)\)(?:, |$)/u.exec(rest)!; + const printed = item[1]!; + items.push({ value: printed.startsWith('"') ? JSON.parse(printed) as MetadataScalar : printed, documents: Number(item[2]) }); + rest = rest.slice(item[0].length); + } + return items; +} + +/** The printed value column of each row: everything before the two-space gap and the count. */ +function printedValues(output: string): string[] { + return output.split("\n").slice(1).filter(line => line !== "").map(line => line.slice(2).replace(/ {2,}\d+$/, "")); +} + +describe("formatMetadataKeySummaries values", () => { + test("prints a plain string bare", () => { + expect(printedValues(formatMetadataKeySummaries(stringResult(["published", "docs team", "v1.2-rc", "é"]), OPTIONS))) + .toEqual(["published", "docs team", "v1.2-rc", "é"]); + }); + + test("quotes every string whose bare form could be read as another value or as layout", () => { + const ambiguous = ["", '""', " padded", "padded ", "two spaces", "tab\tin", "line\nbreak", "42", "-1.5", "true", "null", "[1]", "back\\slash"]; + + const printed = printedValues(formatMetadataKeySummaries(stringResult(ambiguous), OPTIONS)); + + expect(printed.every(value => value.startsWith('"') && value.endsWith('"'))).toBe(true); + expect(printed.map(value => JSON.parse(value) as MetadataScalar)).toEqual(ambiguous); + }); + + test("escapes every control character, including the ones JSON.stringify leaves literal", () => { + const controls = ["bell\u0007", "del\u007f", "csi\u009b", "line\u2028sep", "para\u2029sep", "nul\u0000"]; + + const output = formatMetadataKeySummaries(stringResult(controls), OPTIONS); + + expect(output).not.toMatch(/[\u0000-\u0008\u000b-\u001f\u007f-\u009f\u2028\u2029]/u); + expect(printedValues(output).map(value => JSON.parse(value) as MetadataScalar)).toEqual(controls); + }); + + test("quotes a string that the compact list's delimiters or tails would otherwise absorb", () => { + // "a (1), b" and the three items a, b, c would print identically without quotes. + const values = ["a (1), b", "c", "3 more", "...", "(x)", "plain value"]; + + const output = formatMetadataKeySummaries(splitResult(values, 2), OPTIONS); + + expect(output.split("\n")[1]).toBe(' string 6 docs "a (1), b" (1), c (1), "3 more" (1), "..." (1), "(x)" (1), plain value (1), 2 more'); + expect(compactItems(output)).toEqual([ + ...values.map(value => ({ value, documents: 1 })), + { value: "2 more", documents: -1 }, + ]); + // The same value prints the same way in the vertical layout. + expect(printedValues(formatMetadataKeySummaries(stringResult(values), OPTIONS))).toEqual(['"a (1), b"', "c", '"3 more"', '"..."', '"(x)"', "plain value"]); + }); + + test("prints numbers and booleans bare", () => { + const typeSummary: MetadataKeyTypeSummary = { + type: "number", multiValued: false, documents: 2, distinctValues: 2, + values: [{ value: 1.5, documents: 1 }, { value: -2, documents: 1 }], + remainingValues: 0, range: { min: -2, median: -0.25, max: 1.5 }, collections: ["notes"], + }; + const result: ListMetadataResult = { documents: 2, totalKeys: 1, keys: [{ key: "n", documents: 2, types: [typeSummary] }], remainingKeys: 0 }; + + expect(formatMetadataKeySummaries(result, OPTIONS)).toBe("n number 2 of 2 documents 2 distinct\n min -2 median -0.25 max 1.5\n -2 (1) 1.5 (1)"); + }); +}); + +describe("formatMetadataKeySummaries empty states", () => { + test("keeps the filter line when no entry matched", () => { + const result: ListMetadataResult = { documents: 3, filteredDocuments: 1, totalKeys: 0, keys: [], remainingKeys: 0 }; + + expect(formatMetadataKeySummaries(result, OPTIONS)).toBe("filter: 1 of 3 documents\n\nNo metadata matches."); + }); + + test("keeps the filter line when the key window falls past the end", () => { + const result: ListMetadataResult = { documents: 3, filteredDocuments: 2, totalKeys: 4, keys: [], remainingKeys: 0 }; + + expect(formatMetadataKeySummaries(result, { ...OPTIONS, keyOffset: 10 })).toBe("filter: 2 of 3 documents\n\nNo keys at --key-offset 10, 4 keys in total."); + }); +}); From 2a29f35f660df7b9f47d680d8dd5987f051c85c6 Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Tue, 15 Sep 2026 13:54:39 -0700 Subject: [PATCH 19/82] feat(metadata): surface metadata in collection list, show, and status Makes metadata visible from the commands a user or agent already runs first, so the drill-down is discoverable without knowing it exists: - `qmd collection list` adds a `Metadata:` line per collection naming the top five keys by coverage and counting the rest (`+N more`). Omitted when the collection has no metadata, like `Ignore:`. - `qmd collection show` adds a `Metadata` section: a coverage line (keys, documents with metadata, pending extraction), then the top five keys as aligned rows with type, coverage, distinct count, and a short value preview (top three strings, numeric range and median, boolean counts, or a pointer when types disagree), then a pointer at `qmd collection metadata ` when keys were left out. - `qmd status` adds one summary line under Documents, placed with the existing pending-extraction line so the two read together. - Every view asks the store for exactly what it prints. `show` requests a five-key, three-value window and reads the total and remainder from the result, `list` reads the first five key names and counts the rest, and `status` counts keys without listing them, so none of them grows with the vocabulary. - src/metadata-store.ts gains the light queries these views need: `countDocumentsWithMetadata()` and an optional collection scope on `countDocumentsPendingMetadata()`. The pending-extraction warning that `collection metadata` and the filtered search commands print takes that scope, so it counts the collections the command reads and not the whole index. `status` keeps the index-wide count. Collection list batches all key overviews in one query, including the vocabulary totals, instead of counting and ranking separately for each collection. CLI status counts metadata-bearing documents with existence probes and counts keys from ordinal-zero rows, avoiding array expansion. Type-conflict drill-down hints escape apostrophes in the shell-quoted JSON operand. A POSIX shell round-trip regression ensures metadata keys remain literal data, including command-substitution syntax. Assisted-by: Claude Fable 5.1 via Pi --- src/cli/qmd.ts | 50 +++++++++++++++--- src/metadata-format.ts | 89 ++++++++++++++++++++++++++++++++- src/metadata-store.ts | 31 ++++++++---- test/metadata-cli.test.ts | 57 +++++++++++++++++++++ test/metadata-discovery.test.ts | 49 ++++++++++++++++++ test/metadata-format.test.ts | 19 ++++++- 6 files changed, 275 insertions(+), 20 deletions(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index ed9590658..946385643 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -89,12 +89,15 @@ import { import { syncDocumentMetadata, countDocumentsPendingMetadata, + countDocumentsWithMetadata, + countMetadataKeys, listMetadata, + listMetadataCollectionSummaries, MetadataBindingBudgetError, type ListMetadataOptions, type ListMetadataResult, } from "../metadata-store.js"; -import { formatMetadataKeySummaries } from "../metadata-format.js"; +import { formatMetadataKeySummaries, formatMetadataOverview } from "../metadata-format.js"; import type { DocumentMetadata } from "../metadata.js"; import { parseMetadataFilter, parseMetadataMatch, type MetadataFilter, type MetadataMatch } from "../metadata-filter.js"; import { disposeDefaultLlamaCpp, getDefaultLlamaCpp, setDefaultLlamaCpp, LlamaCpp, withLLMSession, pullModels, DEFAULT_MODEL_CACHE_DIR, resolveEmbedModel, resolveGenerateModel, resolveRerankModel, resolveModels, inspectGgufFile, isDarwinMetalMitigationActive } from "../llm.js"; @@ -582,6 +585,10 @@ async function showStatus(): Promise { if (needsEmbedding > 0) { console.log(` ${c.yellow}Pending: ${needsEmbedding} need embedding${c.reset} (run 'qmd embed')`); } + const metadataKeyCount = countMetadataKeys(db); + if (metadataKeyCount > 0) { + console.log(` Metadata: ${metadataKeyCount} keys across ${countDocumentsWithMetadata(db)} files (explore with 'qmd collection metadata')`); + } const pendingMetadata = countDocumentsPendingMetadata(db); if (pendingMetadata > 0) { console.log(` ${c.yellow}Metadata: ${pendingMetadata} need extraction${c.reset} (run 'qmd update'; excluded from --filter searches)`); @@ -1803,6 +1810,7 @@ function collectionList(): void { } console.log(`${c.bold}Collections (${collections.length}):${c.reset}\n`); + const metadataOverviews = listMetadataCollectionSummaries(db, COLLECTION_LIST_METADATA_KEYS); for (const coll of collections) { const updatedAt = coll.last_modified ? new Date(coll.last_modified) : new Date(); @@ -1819,6 +1827,12 @@ function collectionList(): void { console.log(` ${c.dim}Ignore:${c.reset} ${yamlColl.ignore.join(', ')}`); } console.log(` ${c.dim}Files:${c.reset} ${coll.active_count}`); + const metadataOverview = metadataOverviews.get(coll.name); + if (metadataOverview) { + const shownKeys = metadataOverview.keys.map(overview => overview.key); + const hiddenKeys = metadataOverview.totalKeys - shownKeys.length; + console.log(` ${c.dim}Metadata:${c.reset} ${shownKeys.join(', ')}${hiddenKeys > 0 ? `, +${hiddenKeys} more` : ''}`); + } console.log(` ${c.dim}Updated:${c.reset} ${timeAgo}`); console.log(); } @@ -1826,6 +1840,26 @@ function collectionList(): void { closeDb(); } +/** Key names shown on `collection list`, and keys detailed on `collection show`. */ +const COLLECTION_LIST_METADATA_KEYS = 5; + +// The Metadata section of `collection show`: top keys by coverage with a +// value preview, and a pointer at the drill-down for the rest. +function collectionShowMetadata(name: string): void { + const db = getDb(); + const result = listMetadata(db, { collection: name, keyLimit: COLLECTION_LIST_METADATA_KEYS, valueLimit: 3 }); + const documentsWithMetadata = countDocumentsWithMetadata(db, [name]); + const pendingMetadata = countDocumentsPendingMetadata(db, [name]); + closeDb(); + + console.log(formatMetadataOverview(result, { + documentsWithMetadata, + pendingMetadata, + drillDownHint: `qmd collection metadata ${name}`, + colors: c, + })); +} + /** Canonical --mask, with --glob as the alias OpenClaw and others already pass (#536). */ function collectionGlobFromCli(values: { mask?: unknown; glob?: unknown }): string { const mask = typeof values.mask === "string" && values.mask.length > 0 ? values.mask : undefined; @@ -1937,7 +1971,7 @@ function collectionMetadata(collectionNames: string[], options: ListMetadataOpti } // Discovery always applies the extraction gate, filter or not. - warnPendingMetadata(db); + warnPendingMetadata(db, collectionNames); let result: ListMetadataResult; try { @@ -3000,8 +3034,9 @@ function parseCliPredicateFlag(raw: unknown, predicateFlag: CliPredic // Filtered search excludes documents without current metadata extraction; // tell the user when that makes results incomplete. -function warnPendingMetadata(db: Database): void { - const pendingMetadata = countDocumentsPendingMetadata(db); +/** Warn about documents the extraction gate excludes, within the collections the command reads. */ +function warnPendingMetadata(db: Database, collectionNames: string[]): void { + const pendingMetadata = countDocumentsPendingMetadata(db, collectionNames.length > 0 ? collectionNames : undefined); if (pendingMetadata === 0) return; process.stderr.write(`${c.yellow}Warning: ${pendingMetadata} document(s) lack current metadata extraction and are excluded from filtered results. Run 'qmd update'.${c.reset}\n`); } @@ -3013,7 +3048,7 @@ function search(query: string, opts: OutputOptions): void { // Use default collections if none specified const collectionNames = resolveCollectionFilter(opts.collection, true); - if (opts.filter) warnPendingMetadata(db); + if (opts.filter) warnPendingMetadata(db, collectionNames); // Use large limit for --all, otherwise fetch more than needed and let outputResults filter const fetchLimit = opts.all ? 100000 : Math.max(50, opts.limit * 2); @@ -3064,7 +3099,7 @@ async function vectorSearch(query: string, opts: OutputOptions, _model: string = const collectionNames = resolveCollectionFilter(opts.collection, true); checkIndexHealth(store.db); - if (opts.filter) warnPendingMetadata(store.db); + if (opts.filter) warnPendingMetadata(store.db, collectionNames); await withLLMSession(async () => { let results = await vectorSearchQuery(store, query, { @@ -3109,7 +3144,7 @@ async function querySearch(query: string, opts: OutputOptions, _embedModel: stri const collectionNames = resolveCollectionFilter(opts.collection, true); checkIndexHealth(store.db); - if (opts.filter) warnPendingMetadata(store.db); + if (opts.filter) warnPendingMetadata(store.db, collectionNames); // Check for structured query syntax (lex:/vec:/hyde:/intent: prefixes) const parsed = parseStructuredQuery(query); @@ -4809,6 +4844,7 @@ if (isMain) { const ctxCount = Object.keys(col.context).length; console.log(` Contexts: ${ctxCount}`); } + collectionShowMetadata(name); break; } diff --git a/src/metadata-format.ts b/src/metadata-format.ts index 1d2ba8cc1..148140502 100644 --- a/src/metadata-format.ts +++ b/src/metadata-format.ts @@ -105,6 +105,87 @@ export function formatMetadataKeySummary(summary: MetadataKeySummary, result: Li return lines.join("\n"); } +export interface FormatMetadataOverviewOptions { + /** Active, extracted documents declaring at least one key. */ + documentsWithMetadata: number; + /** Active documents awaiting extraction, mentioned so the coverage reads honestly. */ + pendingMetadata: number; + /** Command that shows the rest, e.g. `qmd collection metadata notes`. */ + drillDownHint: string; + colors?: MetadataFormatColors; +} + +/** + * The `Metadata:` section of `collection show`: a coverage line, then the + * keys in the result's window as aligned rows with a short value preview, + * then a pointer at the drill-down when keys were left out. Indented to sit + * under the other `show` fields. + */ +export function formatMetadataOverview(result: ListMetadataResult, options: FormatMetadataOverviewOptions): string { + const colors = options.colors ?? NO_COLORS; + const pendingNote = options.pendingMetadata > 0 ? ` (${formatCount(options.pendingMetadata)} pending extraction)` : ""; + + if (result.totalKeys === 0) return ` Metadata: none${pendingNote}`; + + const keyLabel = result.totalKeys === 1 ? "key" : "keys"; + const lines = [` Metadata: ${formatCount(result.totalKeys)} ${keyLabel}, ${formatCount(options.documentsWithMetadata)} of ${formatCount(result.documents)} documents${pendingNote}`]; + + const shownKeys = result.keys; + const keyWidth = Math.max(...shownKeys.map(summary => summary.key.length)); + const typeWidth = Math.max(...shownKeys.map(summary => summary.types.map(typeLabelOf).join(" | ").length)); + const documentsWidth = Math.max(...shownKeys.map(summary => formatCount(summary.documents).length)); + const distinctWidth = Math.max(...shownKeys.map(summary => formatCount(distinctValuesOf(summary)).length)); + + for (const summary of shownKeys) { + const typeLabel = summary.types.map(typeLabelOf).join(" | "); + const columns = [ + `${colors.cyan}${summary.key.padEnd(keyWidth)}${colors.reset}`, + `${colors.dim}${typeLabel.padEnd(typeWidth)}${colors.reset}`, + `${formatCount(summary.documents).padStart(documentsWidth)} ${documentsLabelOf(summary.documents)}`, + `${formatCount(distinctValuesOf(summary)).padStart(distinctWidth)} distinct`, + ]; + const preview = formatValuePreview(summary, options.drillDownHint); + if (preview) columns.push(preview); + lines.push(` ${columns.join(" ")}`); + } + + if (result.remainingKeys > 0) { + lines.push(` ${colors.dim}${formatCount(result.remainingKeys)} more ${result.remainingKeys === 1 ? "key" : "keys"}, see '${options.drillDownHint}'${colors.reset}`); + } + + return lines.join("\n"); +} + +/** + * One-line value preview for the overview row. Strings list the window with + * a trailing ellipsis when truncated, and nothing at all when every value is + * unique (a value list would be noise). Numbers give the range, booleans the + * two counts, and a type conflict points at the drill-down. + */ +function formatValuePreview(summary: MetadataKeySummary, drillDownHint: string): string { + if (summary.types.length > 1) { + // JSON quoting does not protect an apostrophe inside a shell single quote. + const match = JSON.stringify({ field: "key", operator: "eq", value: summary.key }).replaceAll("'", "'\\''"); + return `types disagree, see: ${drillDownHint} --match '${match}'`; + } + + const typeSummary = summary.types[0]!; + if (typeSummary.type === "boolean") return formatBooleanCounts(typeSummary.values).replace(" ", ", "); + if (typeSummary.type === "number") { + const range = typeSummary.range!; + return `${formatValue(range.min)} to ${formatValue(range.max)}, median ${formatValue(range.median)}`; + } + if (typeSummary.distinctValues === typeSummary.documents) return ""; + + const preview = typeSummary.values.map(count => `${formatValue(count.value)} (${formatCount(count.documents)})`); + if (typeSummary.remainingValues > 0) preview.push("..."); + return preview.join(", "); +} + +function distinctValuesOf(summary: MetadataKeySummary): number { + return summary.types.reduce((sum, typeSummary) => sum + typeSummary.distinctValues, 0); +} + /** Body for a key with one type: vertical values for strings, a range for numbers, one line for booleans. */ function formatTypeBody(typeSummary: MetadataKeyTypeSummary): string[] { if (typeSummary.type === "boolean") return [` ${formatBooleanCounts(typeSummary.values)}`]; @@ -138,8 +219,7 @@ function formatTypeSplit(typeSummaries: MetadataKeyTypeSummary[], options: Forma if (typeSummary.remainingValues > 0) inlineValues.push(`${formatCount(typeSummary.remainingValues)} more`); valuesSummary = inlineValues.join(", "); } - const documentsLabel = typeSummary.documents === 1 ? "doc " : "docs"; - const row = ` ${typeSummary.type.padEnd(typeWidth)} ${formatCount(typeSummary.documents).padStart(documentsWidth)} ${documentsLabel} ${valuesSummary}`; + const row = ` ${typeSummary.type.padEnd(typeWidth)} ${formatCount(typeSummary.documents).padStart(documentsWidth)} ${documentsLabelOf(typeSummary.documents)} ${valuesSummary}`; return { row, collections: typeSummary.collections.join(", ") }; }); @@ -179,6 +259,11 @@ function formatBooleanCounts(values: MetadataValueCount[]): string { return parts.join(" "); } +/** Padded so `doc` and `docs` rows stay column-aligned. */ +function documentsLabelOf(documents: number): string { + return documents === 1 ? "doc " : "docs"; +} + function formatValue(value: string | number | boolean): string { if (typeof value !== "string") return String(value); return isUnambiguousBare(value) ? value : formatQuoted(value); diff --git a/src/metadata-store.ts b/src/metadata-store.ts index 0bf515498..13777ed4a 100644 --- a/src/metadata-store.ts +++ b/src/metadata-store.ts @@ -175,9 +175,11 @@ function isDocumentMetadataCurrent(db: Database, documentId: number): boolean { /** * Count active documents without a current, error-free metadata extraction. * These documents are excluded from filtered search until `qmd update` runs. + * Scoped to `collectionNames` when given, otherwise the whole index. */ -export function countDocumentsPendingMetadata(db: Database): number { - const row = db.prepare(` +export function countDocumentsPendingMetadata(db: Database, collectionNames?: string[]): number { + const params: SQLiteValue[] = [METADATA_EXTRACTION_VERSION]; + let sql = ` SELECT COUNT(*) as c FROM documents d WHERE d.active = 1 AND NOT EXISTS ( @@ -185,8 +187,12 @@ export function countDocumentsPendingMetadata(db: Database): number { WHERE dm.document_id = d.id AND dm.extraction_version = ? AND dm.extraction_error IS NULL - ) - `).get(METADATA_EXTRACTION_VERSION) as { c: number }; + )`; + if (collectionNames) { + sql += ` AND d.collection IN (SELECT value FROM json_each(?))`; + params.push(JSON.stringify(collectionNames)); + } + const row = db.prepare(sql).get(...params) as { c: number }; return row.c; } @@ -571,17 +577,22 @@ function compareBinary(left: string, right: string): number { /** Distinct metadata keys declared by active, extracted documents in scope. */ export function countMetadataKeys(db: Database, collectionNames?: string[]): number { - return countKeys(db, buildRegion(buildEligibleCte(collectionNames, undefined))); + const region = buildRegion(buildEligibleCte(collectionNames, undefined)); + const row = db.prepare(` + ${region.withSql} + SELECT COUNT(DISTINCT mv.key) AS c ${region.fromSql} WHERE mv.ordinal = 0 + `).get(...region.withParams, ...region.fromParams) as { c: number }; + return row.c; } /** Active, extracted documents in scope that declare at least one metadata key. */ export function countDocumentsWithMetadata(db: Database, collectionNames?: string[]): number { - const region = buildRegion(buildEligibleCte(collectionNames, undefined)); + const eligible = buildEligibleCte(collectionNames, undefined); const row = db.prepare(` - ${region.withSql} - SELECT COUNT(DISTINCT mv.document_id) AS c - ${region.fromSql} - `).get(...region.withParams, ...region.fromParams) as { c: number }; + ${eligible.withSql} + SELECT COUNT(*) AS c FROM eligible e + WHERE EXISTS (SELECT 1 FROM document_metadata_values mv WHERE mv.document_id = e.document_id) + `).get(...eligible.withParams) as { c: number }; return row.c; } diff --git a/test/metadata-cli.test.ts b/test/metadata-cli.test.ts index 2741aa1c2..9eedf50c9 100644 --- a/test/metadata-cli.test.ts +++ b/test/metadata-cli.test.ts @@ -9,6 +9,7 @@ import { tmpdir } from "node:os"; import { join, dirname } from "node:path"; import { fileURLToPath } from "node:url"; import { spawn } from "node:child_process"; +import { openDatabase } from "../src/db.js"; const thisDir = dirname(fileURLToPath(import.meta.url)); const projectRoot = join(thisDir, ".."); @@ -490,3 +491,59 @@ describe("qmd collection metadata", () => { expect(all.stderr).toContain("Use --all-values, --all-keys, or both"); }, 30000); }); + +describe("metadata in collection list, show, and status", () => { + test("collection list names the top keys and counts the rest", async () => { + const { stdout, exitCode } = await runQmd(["collection", "list"]); + expect(exitCode).toBe(0); + expect(stdout).toContain(" Metadata: status, priority, topics, owner, reviewed, +1 more\n"); + }, 30000); + + test("collection show details the top keys and points at the drill-down", async () => { + const { stdout, exitCode } = await runQmd(["collection", "show", "notes"]); + expect(exitCode).toBe(0); + + const metadataSection = stdout.slice(stdout.indexOf(" Metadata:")); + expect(metadataSection).toBe([ + " Metadata: 6 keys, 5 of 6 documents", + " status string 5 docs 3 distinct draft (2), published (2), archived (1)", + ` priority number | string 3 docs 3 distinct types disagree, see: qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"priority"}'`, + " topics string[] 2 docs 13 distinct typescript (2), agents (1), architecture (1), ...", + " owner string 1 doc 1 distinct", + " reviewed boolean 1 doc 1 distinct false 1", + " 1 more key, see 'qmd collection metadata notes'", + "", + ].join("\n")); + }, 30000); + + test("status summarizes metadata and points at the drill-down", async () => { + const { stdout, exitCode } = await runQmd(["status"]); + expect(exitCode).toBe(0); + expect(stdout).toContain(" Metadata: 6 keys across 5 files (explore with 'qmd collection metadata')\n"); + }, 30000); + + test("warns about pending extraction only within the collections the command reads", async () => { + const otherDir = join(testDir, "other"); + await mkdir(otherDir, { recursive: true }); + await writeFile(join(otherDir, "aged.md"), "---\nqmd:\n metadata:\n status: draft\n---\n\n# Aged\n"); + expect((await runQmd(["collection", "add", otherDir, "--name", "other"])).exitCode).toBe(0); + + const ageOther = "UPDATE document_metadata SET extraction_version = 0 WHERE document_id IN (SELECT id FROM documents WHERE collection = 'other')"; + const db = openDatabase(dbPath); + db.prepare(ageOther).run(); + db.close(); + + try { + const notes = await runQmd(["collection", "metadata", "notes"]); + expect(notes.exitCode).toBe(0); + expect(notes.stderr).not.toContain("lack current metadata extraction"); + + const other = await runQmd(["collection", "metadata", "other"]); + expect(other.exitCode).toBe(0); + expect(other.stderr).toContain("Warning: 1 document(s) lack current metadata extraction"); + } finally { + // Leave the index as the other tests expect it. + expect((await runQmd(["collection", "remove", "other"])).exitCode).toBe(0); + } + }, 60000); +}); diff --git a/test/metadata-discovery.test.ts b/test/metadata-discovery.test.ts index 954886c67..35e46bcb6 100644 --- a/test/metadata-discovery.test.ts +++ b/test/metadata-discovery.test.ts @@ -17,6 +17,9 @@ import { type Store, } from "../src/store.js"; import { + countDocumentsPendingMetadata, + countDocumentsWithMetadata, + countMetadataKeys, listMetadata, listMetadataKeys, getMetadataOverview, @@ -926,3 +929,49 @@ describe("listMetadata agrees with filtered search", () => { } }); }); + +describe("status view helpers", () => { + beforeEach(async () => { + await insertMetadataDoc("notes", { status: "published", priority: 3 }); + await insertMetadataDoc("notes", { status: "draft" }); + await insertMetadataDoc("notes", {}); + await insertMetadataDoc("work", { priority: "high", source: "jira" }); + const pendingId = await insertMetadataDoc("work", { status: "ghost" }); + store.db.prepare(`DELETE FROM document_metadata WHERE document_id = ?`).run(pendingId); + }); + + test("listMetadataKeys reports names, coverage, and types in coverage order", () => { + expect(listMetadataKeys(store.db)).toEqual([ + { key: "priority", documents: 2, types: ["number", "string"] }, + { key: "status", documents: 2, types: ["string"] }, + { key: "source", documents: 1, types: ["string"] }, + ]); + expect(listMetadataKeys(store.db, ["work"])).toEqual([ + { key: "priority", documents: 1, types: ["string"] }, + { key: "source", documents: 1, types: ["string"] }, + ]); + expect(listMetadataKeys(store.db, ["missing"])).toEqual([]); + }); + + test("listMetadataKeys windows to the first keys in coverage order", () => { + expect(listMetadataKeys(store.db, undefined, 2).map(overview => overview.key)).toEqual(["priority", "status"]); + expect(listMetadataKeys(store.db, undefined, Infinity)).toHaveLength(3); + }); + + test("countMetadataKeys reports the vocabulary size in scope", () => { + expect(countMetadataKeys(store.db)).toBe(3); + expect(countMetadataKeys(store.db, ["work"])).toBe(2); + expect(countMetadataKeys(store.db, ["missing"])).toBe(0); + }); + + test("countDocumentsWithMetadata counts extracted documents declaring a key", () => { + expect(countDocumentsWithMetadata(store.db)).toBe(3); + expect(countDocumentsWithMetadata(store.db, ["notes"])).toBe(2); + }); + + test("countDocumentsPendingMetadata accepts a collection scope", () => { + expect(countDocumentsPendingMetadata(store.db)).toBe(1); + expect(countDocumentsPendingMetadata(store.db, ["notes"])).toBe(0); + expect(countDocumentsPendingMetadata(store.db, ["work"])).toBe(1); + }); +}); diff --git a/test/metadata-format.test.ts b/test/metadata-format.test.ts index 8cde1ee9b..87250ecca 100644 --- a/test/metadata-format.test.ts +++ b/test/metadata-format.test.ts @@ -4,8 +4,10 @@ * reach (string identity, control characters, empty states under a filter). */ +import * as childProcess from "node:child_process"; + import { describe, test, expect } from "vitest"; -import { formatMetadataKeySummaries, type FormatMetadataOptions } from "../src/metadata-format.js"; +import { formatMetadataKeySummaries, formatMetadataOverview, type FormatMetadataOptions } from "../src/metadata-format.js"; import type { ListMetadataResult, MetadataKeyTypeSummary, MetadataScalar } from "../src/metadata-store.js"; const OPTIONS: FormatMetadataOptions = { @@ -73,6 +75,21 @@ function printedValues(output: string): string[] { return output.split("\n").slice(1).filter(line => line !== "").map(line => line.slice(2).replace(/ {2,}\d+$/, "")); } +test.skipIf(process.platform === "win32")("a type-conflict hint preserves apostrophes and shell syntax in the metadata key", () => { + const result = splitResult(["a", "b"]); + const key = "owner'$(printf injected)'s"; + result.keys[0]!.key = key; + const output = formatMetadataOverview(result, { documentsWithMetadata: result.documents, pendingMetadata: 0, drillDownHint: "qmd collection metadata notes" }); + const command = output.split("types disagree, see: ")[1]!; + + // Stub qmd to capture its arguments. The fixture's substitution is harmless + // if quoting regresses, but it must remain literal data in the JSON operand. + const argumentsText = childProcess.execFileSync("sh", ["-c", `qmd() { printf '%s\\n' "$@"; }\n${command}`], { encoding: "utf8" }); + const args = argumentsText.trimEnd().split("\n"); + expect(args.slice(0, 4)).toEqual(["collection", "metadata", "notes", "--match"]); + expect(JSON.parse(args[4]!)).toEqual({ field: "key", operator: "eq", value: key }); +}); + describe("formatMetadataKeySummaries values", () => { test("prints a plain string bare", () => { expect(printedValues(formatMetadataKeySummaries(stringResult(["published", "docs team", "v1.2-rc", "é"]), OPTIONS))) From 3a583d0b6d9018b517d85857f1cfd48c4e42c308 Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Sat, 12 Sep 2026 15:59:53 -0700 Subject: [PATCH 20/82] feat(metadata): expose metadata discovery on SDK, MCP, and HTTP Ships discovery on the remaining surfaces with the same options object and result shape the CLI uses: - SDK: `listMetadata(options?)` on QMDStore under Collection Management, next to listCollections(). `match` and `filter` are validated at the runtime boundary like the search methods, and the window options by the store, so a caller sees MetadataFilterError, MetadataOptionError, or MetadataBindingBudgetError with the same message every surface prints. Discovery types, MetadataMatch, the error classes, the default limits, the binding budget, and parseMetadataMatch() are exported from the package root, and MetadataValueType now lives in metadata.ts beside the scalar types it describes. - MCP `metadata` tool: flat and read-only, taking `collections`, `match`, `filter` (both validated like the query tool's filter), `keyLimit`, `keyOffset`, `valueLimit`, `valueOffset`, `sort`, and `minCount`. The description teaches the one grammar over two record types (a document for `filter`, a metadata entry for `match`), a table of matches and the question each answers, how to read `totalKeys`, `remainingKeys`, `remainingValues`, and `range`, and how to page. Text content renders the CLI shape through the shared formatter, empty states included so the filter line survives them, naming the option that reaches each remainder; structuredContent is the ListMetadataResult. The schema's integers are safe integers, and a filter and match over the binding budget return a tool error. - MCP `status`: CollectionInfo gains `metadataKeyCount` and a windowed `metadataKeys` (the ten most covered keys with types), mirrored in StatusResult and listed in the text summary with `+N more` and a pointer at the metadata tool, so the call agents make first reveals that metadata exists without growing with the vocabulary. - HTTP `POST /metadata`: same body as the tool. 400 on a non-object body, a non-object or invalid match or filter, a bad sort, a non-array `collections`, a non-number window option, a number outside its domain, or a filter and match over the binding budget (the store's message, verbatim). Returns the ListMetadataResult as its own envelope. Logged like POST /query. Updates the exact tool-list assertions in test/mcp.test.ts and covers each surface in test/metadata-surfaces.test.ts, including paging, the 400 paths, invalid matches and options, unsafe integers, the binding budget on all three surfaces, the filter line over an empty tool result, and status structured content. Full status batches all collection metadata overviews in one query. An internal getStatusSummary path returns only the index facts initialization uses, without pretending uncomputed metadata counts are zero. MCP server creation uses that path, including each stateless HTTP request. The public SDK getStatus result and MCP status tool keep their metadata summaries. Regression tests exercise empty-string negation through SDK, HTTP, and MCP. A real MCP tools/list request succeeds with the metadata value table hidden in the isolated fixture, proving initialization does not read it. Restoring the eager status call makes that test fail. Assisted-by: Claude Fable 5.1 via Pi --- src/index.ts | 63 ++++++- src/mcp/server.ts | 249 +++++++++++++++++++++++++- src/store.ts | 34 +++- test/mcp.test.ts | 4 +- test/metadata-discovery.test.ts | 19 ++ test/metadata-surfaces.test.ts | 307 +++++++++++++++++++++++++++++++- 6 files changed, 668 insertions(+), 8 deletions(-) diff --git a/src/index.ts b/src/index.ts index 5ac03bb34..7bb50d9f2 100644 --- a/src/index.ts +++ b/src/index.ts @@ -73,9 +73,11 @@ import type { MetadataScalar, MetadataScalarArray, MetadataValue, + MetadataValueType, } from "./metadata.js"; import { parseMetadataFilter, + parseMetadataMatch, MetadataFilterError, type MetadataFilter, type MetadataFilterGroup, @@ -84,7 +86,24 @@ import { type MetadataPredicate, type MetadataPredicateGroup, type MetadataPredicateNegation, + type MetadataMatch, + type MetadataEntryCondition, + type MetadataEntryField, } from "./metadata-filter.js"; +import { + listMetadata as storeListMetadata, + MetadataBindingBudgetError, + MetadataOptionError, + DEFAULT_METADATA_KEY_LIMIT, + DEFAULT_METADATA_VALUE_LIMIT, + METADATA_SQL_BINDING_BUDGET, + type ListMetadataOptions, + type ListMetadataResult, + type MetadataKeySummary, + type MetadataKeyTypeSummary, + type MetadataValueCount, + type MetadataKeyOverview, +} from "./metadata-store.js"; import { setConfigSource, loadConfig, @@ -132,6 +151,7 @@ export type { MetadataScalar, MetadataScalarArray, MetadataValue, + MetadataValueType, MetadataFilter, MetadataFilterGroup, MetadataFilterNegation, @@ -139,8 +159,28 @@ export type { MetadataPredicate, MetadataPredicateGroup, MetadataPredicateNegation, + MetadataMatch, + MetadataEntryCondition, + MetadataEntryField, +}; +export { parseMetadataFilter, parseMetadataMatch, MetadataFilterError }; + +// Re-export metadata discovery types (listMetadata() and status metadata keys) +export type { + ListMetadataOptions, + ListMetadataResult, + MetadataKeySummary, + MetadataKeyTypeSummary, + MetadataValueCount, + MetadataKeyOverview, +}; +export { + MetadataBindingBudgetError, + MetadataOptionError, + DEFAULT_METADATA_KEY_LIMIT, + DEFAULT_METADATA_VALUE_LIMIT, + METADATA_SQL_BINDING_BUDGET, }; -export { parseMetadataFilter, MetadataFilterError }; // Re-export the internal Store type for advanced consumers export type { InternalStore }; @@ -309,6 +349,22 @@ export interface QMDStore { /** List all collections with document stats */ listCollections(): Promise<{ name: string; pwd: string; glob_pattern: string; doc_count: number; active_count: number; last_modified: string | null; includeByDefault: boolean }[]>; + /** + * Discover metadata keys, types, and value counts across the documents in + * scope. `filter` selects which documents are counted. `match` selects + * which of their metadata entries are reported, with the filter grammar + * evaluated against each entry (a condition's `field` is the entry's `key` or + * `value`). Keys and values are windowed by `keyLimit`/`keyOffset` and + * `valueLimit`/`valueOffset`, and the result carries the totals and + * remainders needed to page. Every value reported is one an `eq` filter can + * match under the same scope, and the whole result comes from one database + * snapshot. Throws MetadataFilterError for an invalid predicate, + * MetadataOptionError for an option outside its domain, and + * MetadataBindingBudgetError when `filter` and `match` together bind more + * SQL parameters than METADATA_SQL_BINDING_BUDGET. + */ + listMetadata(options?: ListMetadataOptions): Promise; + /** Get names of collections included by default in queries */ getDefaultCollectionNames(): Promise; @@ -516,6 +572,11 @@ export async function createStore(options: StoreOptions): Promise { return result; }, listCollections: async () => storeListCollections(db), + listMetadata: async (opts) => storeListMetadata(db, { + ...opts, + match: opts?.match === undefined ? undefined : parseMetadataMatch(opts.match), + filter: opts?.filter === undefined ? undefined : parseMetadataFilter(opts.filter), + }), getDefaultCollectionNames: async () => { const collections = storeListCollections(db); return collections.filter(c => c.includeByDefault).map(c => c.name); diff --git a/src/mcp/server.ts b/src/mcp/server.ts index e0e05a31e..171622515 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -23,13 +23,20 @@ import { getDefaultDbPath, DEFAULT_MULTI_GET_MAX_BYTES, parseMetadataFilter, + parseMetadataMatch, type QMDStore, type ExpandedQuery, type IndexStatus, type DocumentMetadata, type MetadataFilter, + type MetadataMatch, + type MetadataKeyOverview, + type ListMetadataResult, + MetadataBindingBudgetError, + MetadataOptionError, } from "../index.js"; import { getConfigPath } from "../collections.js"; +import { formatMetadataKeySummaries } from "../metadata-format.js"; import { enableProductionMode } from "../store.js"; import { checkRequestOrigin, resolveOriginGuard } from "./origin-guard.js"; @@ -61,6 +68,28 @@ function validateFilterArgument(filter: unknown): { filter?: MetadataFilter; err } } +/** Same grammar as the filter, validated against metadata entries for discovery. */ +function validateMatchArgument(match: unknown): { match?: MetadataMatch; error?: string } { + if (match === undefined) return {}; + try { + return { match: parseMetadataMatch(match) }; + } catch (err) { + return { error: err instanceof Error ? err.message : String(err) }; + } +} + +/** Pick the named fields of a JSON body that must be numbers when present. */ +function readNumberFields(params: Record, names: string[]): { values: Record; error?: string } { + const values: Record = {}; + for (const name of names) { + const value = params[name]; + if (value === undefined) continue; + if (typeof value !== "number") return { values, error: `Invalid field: ${name} (must be a number)` }; + values[name] = value; + } + return { values }; +} + type StatusResult = { totalDocuments: number; needsEmbedding: number; @@ -71,6 +100,8 @@ type StatusResult = { pattern: string | null; documents: number; lastUpdated: string; + metadataKeyCount: number; + metadataKeys: MetadataKeyOverview[]; }[]; }; @@ -122,7 +153,9 @@ function getPackageVersion(): string { * searchable without a tool call. */ async function buildInstructions(store: QMDStore): Promise { - const status = await store.getStatus(); + // Instructions use index facts, not metadata overviews. Do not rank keys on + // initialization or on each stateless HTTP request just to discard them. + const status = store.internal.getStatusSummary(); const globalCtx = await store.getGlobalContext(); const lines: string[] = []; @@ -610,6 +643,15 @@ Intent-aware lex (C++ performance, not sports): for (const col of status.collections) { summary.push(` - ${col.name}: ${col.path} (${col.documents} docs)`); + if (col.metadataKeys.length > 0) { + const keyLabels = col.metadataKeys.map(overview => `${overview.key} (${overview.types.join(" | ")})`); + const hiddenKeys = col.metadataKeyCount - col.metadataKeys.length; + summary.push(` metadata keys: ${keyLabels.join(", ")}${hiddenKeys > 0 ? `, +${hiddenKeys} more` : ""}`); + } + } + + if (status.collections.some(col => col.metadataKeys.length > 0)) { + summary.push(` Metadata: call the 'metadata' tool to see values and counts before writing a 'filter'`); } return { @@ -619,6 +661,114 @@ Intent-aware lex (C++ performance, not sports): }) ); + // --------------------------------------------------------------------------- + // Tool: qmd_metadata (Metadata discovery) + // --------------------------------------------------------------------------- + + server.registerTool( + "metadata", + { + title: "Metadata Discovery", + description: `Discover which metadata keys exist, what types they hold, and how many documents share each value, so you can write a precise \`filter\` for the query tool. + +Documents carry metadata as \`qmd.metadata\` frontmatter (strings, numbers, booleans, or arrays of one of those). This tool reports what is indexed, never guesses. + +## Mental model + +\`filter\` selects WHICH documents are counted. \`match\` selects WHICH metadata entries of those documents are reported. Both take the same recursive AST as the query tool's \`filter\`. A condition tests one \`field\` of the record under evaluation: for \`filter\` the record is a document and \`field\` names one of its metadata keys, for \`match\` the record is a metadata entry and \`field\` is \`"key"\` (the entry's key name) or \`"value"\` (its value). Every operator applies (eq/ne/gt/gte/lt/lte, in/nin, contains/prefix/suffix, type, and/or/not, caseInsensitive), except \`exists\` and \`all\`, which have no meaning for a single entry. + +| match | Question answered | +|---|---| +| (none) | Which keys exist, with a window of values each | +| \`{"field":"key","operator":"eq","value":"topics"}\` | Everything about one key | +| \`{"field":"key","operator":"prefix","value":"mem-"}\` | A family of keys | +| \`{"field":"key","operator":"in","value":["tags","topics","labels"]}\` | Which of these key names exist | +| \`{"field":"value","operator":"eq","value":"docs-team"}\` | Which keys hold this value (reverse lookup) | +| \`{"field":"value","operator":"prefix","value":"2025-"}\` | Which keys hold values shaped like this | +| \`{"field":"value","operator":"type","value":"boolean"}\` | Which keys hold booleans | +| \`{"operator":"and","operands":[{"field":"key","operator":"eq","value":"priority"},{"field":"value","operator":"gte","value":3}]}\` | Values of one key above a threshold | +| \`{"operator":"and","operands":[{"field":"key","operator":"eq","value":"priority"},{"field":"value","operator":"type","value":"number"}]}\` | The numeric side of a key whose documents disagree on type | + +Compose with and/or/not to ask several of these in one call. Add \`filter\` to any of them to see what remains after narrowing, e.g. the topics among published documents. The result then reports \`filteredDocuments\`, how many documents pass, and every coverage count is measured against that population. + +## Reading the result + +\`totalKeys\` keys have a matching entry. \`keys\` holds one window of them ordered by coverage (\`keyLimit\`, default 50, from \`keyOffset\`), and \`remainingKeys\` says how many follow the window. Each key splits by type. Metadata is validated per document, never across documents, so a key can hold numbers in some files and strings in others within a single collection as easily as across collections. A key with more than one type reports each type separately with its own document count and contributing collections, so you can see how many documents a typed filter would reach. Per type: \`documents\` holding it, \`distinctValues\`, one window of \`values\` with document counts (\`valueLimit\`, default 10, from \`valueOffset\`), and \`remainingValues\` after the window. Both remainders are exact. Numbers also report \`range\` (min, median, max) for writing gt/lt thresholds. Counts are documents, not values: a document with \`topics: [a, b]\` counts once for each. + +## Paging + +Page keys with \`keyOffset\` (next page starts at \`keyOffset + keys.length\`) and values with \`valueOffset\`, which applies to every key in the result and so reads best after \`match\` narrows to one key. Raise a limit instead when the remainder is small. + +Every value reported here can be matched with \`{field: '', operator: 'eq', value}\` under the same collections.`, + annotations: { readOnlyHint: true, openWorldHint: false }, + inputSchema: z.object({ + collections: z.array(z.string()).optional().describe("Restrict to these collections (default: the same collections query searches)"), + match: z.record(z.string(), z.unknown()).optional().describe( + "Report only metadata entries matching this condition. Same recursive AST as 'filter', evaluated per entry: " + + "a condition's field is 'key' (the entry's key name) or 'value' (its value). Default: every entry. " + + "Example: {\"field\":\"key\",\"operator\":\"eq\",\"value\":\"topics\"}" + ), + filter: z.record(z.string(), z.unknown()).optional().describe( + "Count only documents matching this metadata filter. Same recursive AST as the query tool's 'filter'. " + + "Example: {\"field\":\"status\",\"operator\":\"eq\",\"value\":\"published\"}" + ), + keyLimit: z.number().int().positive().optional().default(50).describe("Keys reported (default: 50). 'remainingKeys' says how many follow the window"), + keyOffset: z.number().int().nonnegative().optional().default(0).describe("Keys skipped before the window, in report order (default: 0)"), + valueLimit: z.number().int().positive().optional().default(10).describe("Values reported per key and type (default: 10). 'remainingValues' says how many follow the window"), + valueOffset: z.number().int().nonnegative().optional().default(0).describe("Values skipped per key and type before the window, in 'sort' order (default: 0)"), + sort: z.enum(["count", "value"]).optional().default("count").describe("Order values by document count descending (default) or by value ascending"), + minCount: z.number().int().positive().optional().default(1).describe("Hide values held by fewer documents than this (default: 1)"), + }), + }, + track(async ({ collections, match, filter, keyLimit, keyOffset, valueLimit, valueOffset, sort, minCount }) => { + const matchValidation = validateMatchArgument(match); + const filterValidation = validateFilterArgument(filter); + const validationError = matchValidation.error ?? filterValidation.error; + if (validationError) { + return { + content: [{ type: "text" as const, text: `Error: ${validationError}` }], + isError: true, + }; + } + + const effectiveCollections = collections ?? defaultCollectionNames; + let result: ListMetadataResult; + try { + result = await store.listMetadata({ + collection: effectiveCollections.length > 0 ? effectiveCollections : undefined, + match: matchValidation.match, + filter: filterValidation.filter, + keyLimit, + keyOffset, + valueLimit, + valueOffset, + sort, + minCount, + }); + } catch (err) { + if (!(err instanceof MetadataBindingBudgetError)) throw err; + return { + content: [{ type: "text" as const, text: `Error: ${err.message}` }], + isError: true, + }; + } + + const text = formatMetadataKeySummaries(result, { + showCollections: effectiveCollections.length !== 1, + valueWindowHint: "a higher 'valueLimit' or a 'valueOffset'", + keyWindowHint: "a higher 'keyLimit' or a 'keyOffset'", + keyOffset, + keyOffsetLabel: "keyOffset", + emptyMessage: "No metadata matches. Call without match/filter to see every key, or check the status tool for collections with metadata.", + }); + + return { + content: [{ type: "text", text }], + structuredContent: result, + }; + }) + ); + return server; } @@ -1105,6 +1255,103 @@ export async function startMcpHttpServer( return; } + // REST endpoint: POST /metadata — metadata discovery, same body as the metadata tool + if (pathname === "/metadata" && nodeReq.method === "POST") { + const rawBody = await collectBody(nodeReq); + let parsedParams: unknown; + try { + parsedParams = rawBody.trim() === "" ? {} : JSON.parse(rawBody); + } catch { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: "Invalid JSON body" })); + return; + } + if (typeof parsedParams !== "object" || parsedParams === null || Array.isArray(parsedParams)) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: "JSON body must be an object" })); + return; + } + const params = parsedParams as Record; + + // Optional metadata filter — must be an object and a valid filter AST + let restFilter: MetadataFilter | undefined; + if (params.filter !== undefined) { + if (typeof params.filter !== "object" || params.filter === null || Array.isArray(params.filter)) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: "Invalid field: filter (must be an object)" })); + return; + } + const filterValidation = validateFilterArgument(params.filter); + if (filterValidation.error) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: filterValidation.error })); + return; + } + restFilter = filterValidation.filter; + } + + // Optional metadata match, validated the same way against entries + let restMatch: MetadataMatch | undefined; + if (params.match !== undefined) { + if (typeof params.match !== "object" || params.match === null || Array.isArray(params.match)) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: "Invalid field: match (must be an object)" })); + return; + } + const matchValidation = validateMatchArgument(params.match); + if (matchValidation.error) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: matchValidation.error })); + return; + } + restMatch = matchValidation.match; + } + + if (params.sort !== undefined && params.sort !== "count" && params.sort !== "value") { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: "Invalid field: sort (must be 'count' or 'value')" })); + return; + } + + if (params.collections !== undefined && !Array.isArray(params.collections)) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: "Invalid field: collections (must be an array)" })); + return; + } + + // The JSON type is checked here; the store checks each number's domain. + const numberFields = readNumberFields(params, ["keyLimit", "keyOffset", "valueLimit", "valueOffset", "minCount"]); + if (numberFields.error) { + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: numberFields.error })); + return; + } + + // Use default collections if none specified + const effectiveCollections = params.collections ? params.collections.map(String) : defaultCollectionNames; + + let result: ListMetadataResult; + try { + result = await store.listMetadata({ + collection: effectiveCollections.length > 0 ? effectiveCollections : undefined, + match: restMatch, + filter: restFilter, + sort: params.sort, + ...numberFields.values, + }); + } catch (err) { + if (!(err instanceof MetadataOptionError) && !(err instanceof MetadataBindingBudgetError)) throw err; + nodeRes.writeHead(400, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify({ error: err.message })); + return; + } + + nodeRes.writeHead(200, { "Content-Type": "application/json" }); + nodeRes.end(JSON.stringify(result)); + log(`${ts()} POST /metadata ${result.keys.length} keys (${Date.now() - reqStart}ms)`); + return; + } + if (pathname === "/mcp") { const rawBody = nodeReq.method !== "GET" && nodeReq.method !== "HEAD" ? await collectBody(nodeReq) diff --git a/src/store.ts b/src/store.ts index 936d0f61c..2c719da1e 100644 --- a/src/store.ts +++ b/src/store.ts @@ -44,7 +44,9 @@ import { syncDocumentMetadata, countDocumentsPendingMetadata, getMetadataByFilepath, + listMetadataCollectionSummaries, parseMetadataJson, + type MetadataKeyOverview, } from "./metadata-store.js"; // ============================================================================= @@ -1511,6 +1513,7 @@ export type Store = { getHashesNeedingEmbedding: (model?: string) => number; getIndexHealth: (model?: string) => IndexHealthInfo; getStatus: (model?: string) => IndexStatus; + getStatusSummary: (model?: string) => IndexStatusSummary; // Caching getCacheKey: typeof getCacheKey; @@ -2265,6 +2268,7 @@ export function createStore(dbPath?: string): Store { getHashesNeedingEmbedding: (model?: string) => getHashesNeedingEmbedding(db, undefined, model ?? store.llm?.embedModelName ?? DEFAULT_EMBED_MODEL), getIndexHealth: (model?: string) => getIndexHealth(db, model ?? store.llm?.embedModelName ?? DEFAULT_EMBED_MODEL), getStatus: (model?: string) => getStatus(db, model ?? store.llm?.embedModelName ?? DEFAULT_EMBED_MODEL), + getStatusSummary: (model?: string) => getStatusSummary(db, model ?? store.llm?.embedModelName ?? DEFAULT_EMBED_MODEL), // Caching getCacheKey, @@ -2528,8 +2532,19 @@ export type CollectionInfo = { pattern: string | null; documents: number; lastUpdated: string; + /** Distinct metadata keys declared in this collection. */ + metadataKeyCount: number; + /** + * The most covered metadata keys in this collection, with coverage and + * types, by coverage. Windowed to STATUS_METADATA_KEY_LIMIT; `listMetadata` + * pages through the rest. + */ + metadataKeys: MetadataKeyOverview[]; }; +/** Keys a status view names per collection before pointing at discovery. */ +export const STATUS_METADATA_KEY_LIMIT = 10; + export type IndexStatus = { totalDocuments: number; needsEmbedding: number; @@ -2539,6 +2554,11 @@ export type IndexStatus = { collections: CollectionInfo[]; }; +/** Index facts for initialization, without computing unused metadata overviews. */ +export type IndexStatusSummary = Omit & { + collections: Omit[]; +}; + // ============================================================================= // Index health // ============================================================================= @@ -5238,6 +5258,18 @@ export function findDocuments( // ============================================================================= export function getStatus(db: Database, model: string = DEFAULT_EMBED_MODEL): IndexStatus { + const statusSummary = getStatusSummary(db, model); + const metadataOverviews = listMetadataCollectionSummaries(db, STATUS_METADATA_KEY_LIMIT); + return { + ...statusSummary, + collections: statusSummary.collections.map(collection => { + const metadataOverview = metadataOverviews.get(collection.name); + return { ...collection, metadataKeyCount: metadataOverview?.totalKeys ?? 0, metadataKeys: metadataOverview?.keys ?? [] }; + }), + }; +} + +export function getStatusSummary(db: Database, model: string = DEFAULT_EMBED_MODEL): IndexStatusSummary { // DB is source of truth for collections — config provides supplementary metadata const dbCollections = db.prepare(` SELECT @@ -5253,7 +5285,7 @@ export function getStatus(db: Database, model: string = DEFAULT_EMBED_MODEL): In const storeCollections = getStoreCollections(db); const configLookup = new Map(storeCollections.map(c => [c.name, { path: c.path, pattern: c.pattern }])); - const collections: CollectionInfo[] = dbCollections.map(row => { + const collections: IndexStatusSummary["collections"] = dbCollections.map(row => { const config = configLookup.get(row.name); return { name: row.name, diff --git a/test/mcp.test.ts b/test/mcp.test.ts index dc0e4c523..ecb5f9664 100644 --- a/test/mcp.test.ts +++ b/test/mcp.test.ts @@ -1131,7 +1131,7 @@ describe.skipIf(!!process.env.CI)("MCP HTTP Transport", () => { expect(headers.get("mcp-session-id")).toBeNull(); const toolNames = json.result.tools.map((t: any) => t.name); - expect(toolNames).toEqual(["query", "get", "multi_get", "status"]); + expect(toolNames).toEqual(["query", "get", "multi_get", "status", "metadata"]); }); test("POST /mcp tools/call query returns results", async () => { @@ -1309,7 +1309,7 @@ describe("MCP HTTP Transport — 2026-07-28 protocol", () => { expect(json.result.ttlMs).toBe(60_000); expect(json.result.cacheScope).toBe("private"); const toolNames = json.result.tools.map((t: { name: string }) => t.name); - expect(toolNames).toEqual(["query", "get", "multi_get", "status"]); + expect(toolNames).toEqual(["query", "get", "multi_get", "status", "metadata"]); const serverInfo = json.result._meta?.["io.modelcontextprotocol/serverInfo"]; expect(serverInfo?.name).toBe("qmd"); }); diff --git a/test/metadata-discovery.test.ts b/test/metadata-discovery.test.ts index 35e46bcb6..01f13d786 100644 --- a/test/metadata-discovery.test.ts +++ b/test/metadata-discovery.test.ts @@ -14,6 +14,8 @@ import { insertDocument, hashContent, searchFTS, + getStatus, + getStatusSummary, type Store, } from "../src/store.js"; import { @@ -969,6 +971,23 @@ describe("status view helpers", () => { expect(countDocumentsWithMetadata(store.db, ["notes"])).toBe(2); }); + test("status batches metadata overviews and initialization reads no metadata values", () => { + const statements: string[] = []; + const observed = observeStatements(store.db, sql => statements.push(sql)); + const status = getStatus(observed); + expect(status.collections).toHaveLength(2); + expect(status.collections.find(collection => collection.name === "notes")?.metadataKeyCount).toBe(2); + expect(statements.filter(sql => sql.includes("document_metadata_values"))).toHaveLength(1); + + statements.length = 0; + const summary = getStatusSummary(observed); + expect(summary.totalDocuments).toBe(status.totalDocuments); + expect(summary.pendingMetadata).toBe(status.pendingMetadata); + expect(summary.collections.map(collection => collection.name)).toEqual(status.collections.map(collection => collection.name)); + expect(summary.collections[0]).not.toHaveProperty("metadataKeys"); + expect(statements.some(sql => sql.includes("document_metadata_values"))).toBe(false); + }); + test("countDocumentsPendingMetadata accepts a collection scope", () => { expect(countDocumentsPendingMetadata(store.db)).toBe(1); expect(countDocumentsPendingMetadata(store.db, ["notes"])).toBe(0); diff --git a/test/metadata-surfaces.test.ts b/test/metadata-surfaces.test.ts index 14ff10eac..99f89163e 100644 --- a/test/metadata-surfaces.test.ts +++ b/test/metadata-surfaces.test.ts @@ -10,7 +10,14 @@ import { unlinkSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import YAML from "yaml"; -import { createStore, type QMDStore } from "../src/index.js"; +import { + createStore, + MetadataBindingBudgetError, + MetadataOptionError, + type MetadataFilter, + type MetadataMatch, + type QMDStore, +} from "../src/index.js"; import { createStore as createInternalStore, insertContent, @@ -38,6 +45,16 @@ function buildDoc(status: string, body: string): string { return `---\nqmd:\n metadata:\n status: ${status}\n topics: [typescript]\n---\n\n${body}`; } +/** A filter and a match that each pass the parser's limits and together exceed the SQL binding budget. */ +function buildWidePredicates(): { filter: MetadataFilter; match: MetadataMatch } { + const members = Array.from({ length: 64 }, (_, index) => `v${index}`); + const groups = (operand: () => Operand) => Array.from({ length: 7 }, () => Array.from({ length: 31 }, operand)); + return { + filter: { operator: "and", operands: groups(() => ({ field: "status", operator: "all" as const, value: members })).map(operands => ({ operator: "and" as const, operands })) }, + match: { operator: "or", operands: groups(() => ({ field: "value" as const, operator: "in" as const, value: members })).map(operands => ({ operator: "or" as const, operands })) }, + }; +} + // ============================================================================= // SDK // ============================================================================= @@ -94,6 +111,80 @@ describe("SDK metadata filter", () => { filter: { operator: "and", operands: [] }, })).rejects.toThrow(/non-empty 'operands'/); }); + + test("listMetadata summarizes keys, types, and value counts", async () => { + const result = await store.listMetadata(); + expect(result.documents).toBe(2); + expect(result.filteredDocuments).toBeUndefined(); + expect(result.keys.map(summary => summary.key)).toEqual(["status", "topics"]); + + const topics = result.keys[1]!.types[0]!; + expect(topics).toMatchObject({ type: "string", documents: 2, distinctValues: 1, remainingValues: 0, collections: ["docs"] }); + expect(result).toMatchObject({ totalKeys: 2, remainingKeys: 0 }); + expect(topics.values).toEqual([{ value: "typescript", documents: 2 }]); + }); + + test("listMetadata scopes, narrows by filter, and selects entries by match", async () => { + const filtered = await store.listMetadata({ + collection: "docs", + match: { field: "key", operator: "eq", value: "status" }, + filter: { field: "status", operator: "eq", value: "published" }, + }); + expect(filtered.filteredDocuments).toBe(1); + expect(filtered.keys[0]!.types[0]!.values).toEqual([{ value: "published", documents: 1 }]); + + const reverse = await store.listMetadata({ match: { field: "value", operator: "prefix", value: "dra" } }); + expect(reverse.keys.map(summary => summary.key)).toEqual(["status"]); + + const scoped = await store.listMetadata({ collection: "missing" }); + expect(scoped).toEqual({ documents: 0, totalKeys: 0, keys: [], remainingKeys: 0 }); + }); + + test("listMetadata windows keys and values and rejects options outside their domain", async () => { + const page = await store.listMetadata({ keyLimit: 1, keyOffset: 1 }); + expect(page.keys.map(summary => summary.key)).toEqual(["topics"]); + expect(page).toMatchObject({ totalKeys: 2, remainingKeys: 0 }); + + await expect(store.listMetadata({ keyLimit: 0 })).rejects.toThrow(/^Invalid keyLimit: expected a positive integer or Infinity, received 0$/); + await expect(store.listMetadata({ valueOffset: -1 })).rejects.toThrow(MetadataOptionError); + await expect(store.listMetadata({ keyLimit: 1e30 })).rejects.toThrow(MetadataOptionError); + await expect(store.listMetadata(buildWidePredicates())).rejects.toThrow(MetadataBindingBudgetError); + }); + + test("listMetadata validates filters and matches at the SDK runtime boundary", async () => { + // Plain-JS callers bypass the declarations, so the SDK must reject a bad AST at runtime. + const untrustedFilter: import("../src/index.js").ListMetadataOptions = JSON.parse('{"filter":{"field":"status","operator":"equal","value":"x"}}'); + await expect(store.listMetadata(untrustedFilter)).rejects.toThrow(/unknown operator 'equal'/); + + const untrustedMatch: import("../src/index.js").ListMetadataOptions = JSON.parse('{"match":{"field":"status","operator":"eq","value":"x"}}'); + await expect(store.listMetadata(untrustedMatch)).rejects.toThrow(/Invalid metadata match at \$: 'status' is not a field of a metadata entry/); + }); + + test("negated text matches preserve extracted empty strings through the SDK", async () => { + const document = store.internal.db.prepare("SELECT id FROM documents WHERE path = 'published.md'").get() as { id: number }; + replaceDocumentMetadata(store.internal.db, document.id, { metadata: { status: "", topics: ["typescript"] }, extractionVersion: METADATA_EXTRACTION_VERSION }); + try { + for (const operator of ["prefix", "suffix"] as const) { + const result = await store.listMetadata({ match: { operator: "and", operands: [ + { field: "key", operator: "eq", value: "status" }, + { operator: "not", operand: { field: "value", operator, value: "zzz" } }, + ] } }); + expect(result.keys[0]!.documents).toBe(2); + expect(result.keys[0]!.types[0]!.values).toEqual([{ value: "", documents: 1 }, { value: "draft", documents: 1 }]); + } + } finally { + replaceDocumentMetadata(store.internal.db, document.id, { metadata: { status: "published", topics: ["typescript"] }, extractionVersion: METADATA_EXTRACTION_VERSION }); + } + }); + + test("getStatus lists metadata keys per collection", async () => { + const status = await store.getStatus(); + expect(status.collections[0]!.metadataKeyCount).toBe(2); + expect(status.collections[0]!.metadataKeys).toEqual([ + { key: "status", documents: 2, types: ["string"] }, + { key: "topics", documents: 2, types: ["string"] }, + ]); + }); }); // ============================================================================= @@ -121,6 +212,7 @@ describe("MCP and HTTP metadata filter", () => { const internal = createInternalStore(dbPath); await seedDoc(internal.db, "published.md", "# Pub\n\nhttp keyword body", { status: "published" }); await seedDoc(internal.db, "draft.md", "# Draft\n\nhttp keyword body", { status: "draft" }); + await seedDoc(internal.db, "tagged.md", "# Tagged\n\nhttp keyword body", { status: "archived", topics: ["sqlite", "search"], priority: 3 }); const testConfig: CollectionConfig = { collections: { docs: { path: "/test/docs", pattern: "**/*.md" } }, @@ -160,6 +252,10 @@ describe("MCP and HTTP metadata filter", () => { } async function callQueryTool(args: Record): Promise<{ status: number; json: any }> { + return callTool("query", args); + } + + async function callTool(name: string, args: Record): Promise<{ status: number; json: any }> { const res = await fetch(`${baseUrl}/mcp`, { method: "POST", headers: { @@ -167,14 +263,14 @@ describe("MCP and HTTP metadata filter", () => { "Accept": "application/json, text/event-stream", "MCP-Protocol-Version": "2026-07-28", "Mcp-Method": "tools/call", - "Mcp-Name": "query", + "Mcp-Name": name, }, body: JSON.stringify({ jsonrpc: "2.0", id: 1, method: "tools/call", params: { - name: "query", + name, arguments: args, _meta: { "io.modelcontextprotocol/protocolVersion": "2026-07-28", @@ -187,6 +283,55 @@ describe("MCP and HTTP metadata filter", () => { return { status: res.status, json: await res.json() }; } + test("MCP initialization lists tools without reading metadata value rows", async () => { + const internal = createInternalStore(dbPath); + // Make value-table reads fail deterministically instead of testing a noisy + // timing threshold. The running server already has its database connection. + internal.db.exec("ALTER TABLE document_metadata_values RENAME TO hidden_metadata_values"); + try { + const response = await fetch(`${baseUrl}/mcp`, { + method: "POST", + headers: { "Content-Type": "application/json", "Accept": "application/json, text/event-stream", "MCP-Protocol-Version": "2026-07-28", "Mcp-Method": "tools/list" }, + body: JSON.stringify({ jsonrpc: "2.0", id: 1, method: "tools/list", params: { _meta: { + "io.modelcontextprotocol/protocolVersion": "2026-07-28", + "io.modelcontextprotocol/clientInfo": { name: "metadata-test", version: "1.0.0" }, + "io.modelcontextprotocol/clientCapabilities": {}, + } } }), + }); + expect(response.status).toBe(200); + const body = await response.json() as { result: { tools: { name: string }[] } }; + expect(body.result.tools.map(tool => tool.name)).toContain("metadata"); + } finally { + internal.db.exec("ALTER TABLE hidden_metadata_values RENAME TO document_metadata_values"); + internal.close(); + } + }); + + test("HTTP and MCP negated text matches include empty strings", async () => { + const internal = createInternalStore(dbPath); + const document = internal.db.prepare("SELECT id FROM documents WHERE path = 'draft.md'").get() as { id: number }; + replaceDocumentMetadata(internal.db, document.id, { metadata: { status: "" }, extractionVersion: METADATA_EXTRACTION_VERSION }); + try { + for (const operator of ["prefix", "suffix"] as const) { + const args = { match: { operator: "and", operands: [ + { field: "key", operator: "eq", value: "status" }, + { operator: "not", operand: { field: "value", operator, value: "zzz" } }, + ] } }; + const http = await postJson("/metadata", args); + const mcp = await callTool("metadata", args); + expect(http.status).toBe(200); + expect(mcp.status).toBe(200); + expect(mcp.json.result.structuredContent).toEqual(http.json); + expect(http.json.keys[0].documents).toBe(3); + expect(http.json.keys[0].types[0].values).toContainEqual({ value: "", documents: 1 }); + expect(mcp.json.result.content[0].text).toContain('""'); + } + } finally { + replaceDocumentMetadata(internal.db, document.id, { metadata: { status: "draft" }, extractionVersion: METADATA_EXTRACTION_VERSION }); + internal.close(); + } + }); + test("POST /query applies the filter and includes metadata", async () => { const { status, json } = await postJson("/query", { searches: [{ type: "lex", query: "http keyword" }], @@ -256,4 +401,160 @@ describe("MCP and HTTP metadata filter", () => { expect(json.result.isError).toBe(true); expect(json.result.content[0].text).toMatch(/non-empty 'operands'/); }); + + test("MCP metadata tool returns key summaries as structured content and CLI-shaped text", async () => { + const { status, json } = await callTool("metadata", { match: { field: "key", operator: "eq", value: "status" } }); + expect(status).toBe(200); + expect(json.result.isError).toBeFalsy(); + + const result = json.result.structuredContent; + expect(result.documents).toBe(3); + expect(result.keys).toHaveLength(1); + expect(result.keys[0].types[0]).toMatchObject({ type: "string", documents: 3, distinctValues: 3, remainingValues: 0 }); + expect(result.keys[0].types[0].values).toEqual([ + { value: "archived", documents: 1 }, + { value: "draft", documents: 1 }, + { value: "published", documents: 1 }, + ]); + expect(json.result.content[0].text).toBe("status string 3 of 3 documents 3 distinct\n archived 1\n draft 1\n published 1"); + }); + + test("MCP metadata tool narrows by filter, windows values, and reports the remainder", async () => { + const { json } = await callTool("metadata", { + match: { field: "key", operator: "eq", value: "topics" }, + valueLimit: 1, + filter: { field: "status", operator: "eq", value: "archived" }, + }); + const result = json.result.structuredContent; + expect(result.filteredDocuments).toBe(1); + const topics = result.keys[0].types[0]; + expect(topics.multiValued).toBe(true); + expect(topics.values).toHaveLength(1); + expect(topics.remainingValues).toBe(1); + expect(json.result.content[0].text).toContain("filter: 1 of 3 documents\n\ntopics string[] 1 of 1 documents 2 distinct"); + expect(json.result.content[0].text).toContain("1 more value, use a higher 'valueLimit' or a 'valueOffset'"); + }); + + test("MCP metadata tool windows and pages keys", async () => { + const firstPage = await callTool("metadata", { keyLimit: 2 }); + const first = firstPage.json.result.structuredContent; + expect(first.keys.map((summary: { key: string }) => summary.key)).toEqual(["status", "priority"]); + expect(first).toMatchObject({ totalKeys: 3, remainingKeys: 1 }); + expect(firstPage.json.result.content[0].text).toContain("1 more key, use a higher 'keyLimit' or a 'keyOffset'"); + + const secondPage = await callTool("metadata", { keyLimit: 2, keyOffset: 2 }); + expect(secondPage.json.result.structuredContent.keys.map((summary: { key: string }) => summary.key)).toEqual(["topics"]); + expect(secondPage.json.result.structuredContent.remainingKeys).toBe(0); + + const pastTheEnd = await callTool("metadata", { keyOffset: 3 }); + expect(pastTheEnd.json.result.isError).toBeFalsy(); + expect(pastTheEnd.json.result.content[0].text).toBe("No keys at keyOffset 3, 3 keys in total."); + + const badOffset = await callTool("metadata", { keyOffset: -1 }); + expect(badOffset.json.error ?? badOffset.json.result?.isError).toBeTruthy(); + + const unsafeLimit = await callTool("metadata", { keyLimit: 1e30 }); + expect(unsafeLimit.json.error ?? unsafeLimit.json.result?.isError).toBeTruthy(); + }); + + test("MCP metadata tool keeps the filter line over an empty result and reports the binding budget", async () => { + // Every seeded document declares a status, so this filter admits none. + const empty = await callTool("metadata", { filter: { field: "status", operator: "exists", value: false } }); + expect(empty.json.result.isError).toBeFalsy(); + expect(empty.json.result.structuredContent).toMatchObject({ documents: 3, filteredDocuments: 0, totalKeys: 0 }); + expect(empty.json.result.content[0].text).toBe( + "filter: 0 of 3 documents\n\nNo metadata matches. Call without match/filter to see every key, or check the status tool for collections with metadata.", + ); + + const overBudget = await callTool("metadata", buildWidePredicates()); + expect(overBudget.json.result.isError).toBe(true); + expect(overBudget.json.result.content[0].text).toMatch(/^Error: filter and match together bind \d+ SQL parameters, over the budget of 30000\./); + }); + + test("MCP metadata tool rejects invalid filters and matches", async () => { + const { json } = await callTool("metadata", { filter: { field: "status", operator: "equal", value: "x" } }); + expect(json.result.isError).toBe(true); + expect(json.result.content[0].text).toMatch(/unknown operator 'equal'/); + + const badMatch = await callTool("metadata", { match: { field: "value", operator: "all", value: ["x"] } }); + expect(badMatch.json.result.isError).toBe(true); + expect(badMatch.json.result.content[0].text).toMatch(/Invalid metadata match at \$: 'all' has no meaning for a single metadata entry/); + }); + + test("MCP status tool lists metadata keys per collection", async () => { + const { json } = await callTool("status", {}); + expect(json.result.isError).toBeFalsy(); + expect(json.result.structuredContent.collections[0].metadataKeyCount).toBe(3); + expect(json.result.structuredContent.collections[0].metadataKeys).toEqual([ + { key: "status", documents: 3, types: ["string"] }, + { key: "priority", documents: 1, types: ["number"] }, + { key: "topics", documents: 1, types: ["string"] }, + ]); + expect(json.result.content[0].text).toContain("metadata keys: status (string), priority (number), topics (string)"); + expect(json.result.content[0].text).toContain("call the 'metadata' tool"); + }); + + test("POST /metadata returns the same result as the tool", async () => { + const { status, json } = await postJson("/metadata", { match: { field: "value", operator: "type", value: "number" } }); + expect(status).toBe(200); + expect(json.documents).toBe(3); + expect(json.keys.map((summary: { key: string }) => summary.key)).toEqual(["priority"]); + expect(json.keys[0].types[0]).toMatchObject({ type: "number", range: { min: 3, median: 3, max: 3 } }); + + const reverse = await postJson("/metadata", { match: { field: "value", operator: "eq", value: "search" }, valueLimit: 5 }); + expect(reverse.json.keys.map((summary: { key: string }) => summary.key)).toEqual(["topics"]); + + const page = await postJson("/metadata", { keyLimit: 1, keyOffset: 1 }); + expect(page.json.keys.map((summary: { key: string }) => summary.key)).toEqual(["priority"]); + expect(page.json).toMatchObject({ totalKeys: 3, remainingKeys: 1 }); + }); + + test("POST /metadata rejects invalid bodies, filters, matches, and sort with 400", async () => { + const stringFilter = await postJson("/metadata", { filter: "status = published" }); + expect(stringFilter.status).toBe(400); + expect(stringFilter.json.error).toMatch(/must be an object/); + + const stringMatch = await postJson("/metadata", { match: "key = topics" }); + expect(stringMatch.status).toBe(400); + expect(stringMatch.json.error).toMatch(/Invalid field: match \(must be an object\)/); + + const invalidMatch = await postJson("/metadata", { match: { field: "topics", operator: "eq", value: "x" } }); + expect(invalidMatch.status).toBe(400); + expect(invalidMatch.json.error).toMatch(/'topics' is not a field of a metadata entry/); + + const invalidAst = await postJson("/metadata", { filter: { field: "status", operator: "equal", value: "x" } }); + expect(invalidAst.status).toBe(400); + expect(invalidAst.json.error).toMatch(/unknown operator 'equal'/); + + const badSort = await postJson("/metadata", { sort: "size" }); + expect(badSort.status).toBe(400); + expect(badSort.json.error).toMatch(/sort/); + + const stringLimit = await postJson("/metadata", { keyLimit: "5" }); + expect(stringLimit.status).toBe(400); + expect(stringLimit.json.error).toBe("Invalid field: keyLimit (must be a number)"); + + const zeroLimit = await postJson("/metadata", { valueLimit: 0 }); + expect(zeroLimit.status).toBe(400); + expect(zeroLimit.json.error).toBe("Invalid valueLimit: expected a positive integer or Infinity, received 0"); + + const fractionalOffset = await postJson("/metadata", { keyOffset: 1.5 }); + expect(fractionalOffset.status).toBe(400); + expect(fractionalOffset.json.error).toBe("Invalid keyOffset: expected a non-negative integer, received 1.5"); + + const stringCollections = await postJson("/metadata", { collections: "docs" }); + expect(stringCollections.status).toBe(400); + expect(stringCollections.json.error).toBe("Invalid field: collections (must be an array)"); + + const unsafeLimit = await postJson("/metadata", { keyLimit: 1e30 }); + expect(unsafeLimit.status).toBe(400); + expect(unsafeLimit.json.error).toBe("Invalid keyLimit: expected a positive integer or Infinity, received 1e+30"); + + const overBudget = await postJson("/metadata", buildWidePredicates()); + expect(overBudget.status).toBe(400); + expect(overBudget.json.error).toMatch(/^filter and match together bind \d+ SQL parameters, over the budget of 30000\./); + + const arrayBody = await fetch(`${baseUrl}/metadata`, { method: "POST", headers: { "Content-Type": "application/json" }, body: "[]" }); + expect(arrayBody.status).toBe(400); + }); }); From f32848b7a736a693c1abb5c708a2e072a65680c2 Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Sat, 12 Sep 2026 16:29:07 -0700 Subject: [PATCH 21/82] docs(metadata): document metadata discovery - README: a "Metadata Discovery" subsection after "Metadata Filtering" with the mental model (same metadata, different unit: filter narrows documents, match narrows the metadata itself, one predicate grammar over both), a table of matches and the question each answers, a wide-to-narrow walkthrough ending in a validated query, exact CLI output for string, number, boolean, threshold, filtered, and type-conflict keys, the two windows, how they page, and what they do and do not bound, the reading rules (including when a string prints quoted and that a result is one snapshot), and the SDK method with its predicate, option, and error types, MCP tool, and HTTP route. Also the new subcommand in the collection commands block, the `metadata` tool parameters, and the `POST /metadata` endpoint. - CHANGELOG: one entry under Unreleased / Added. - skills/qmd/SKILL.md: a "Discover metadata before filtering" section written for an agent deciding what to filter on: the match shapes worth knowing, how to read the header and the filter line, the window footers and how to page past them, numeric ranges, and type splits. - CLAUDE.md: `qmd collection metadata` in the command table and the Collection Management block. Examples use the status, topics, priority, and reviewed keys the filtering docs already establish, so the two sections read as one. Qualify the windows as bounds on key/value lists, not query work or collection provenance. Document the current-data cost of status summaries and their exclusion from MCP initialization. Name the filter target field consistently and explain explicit rejection of unsupported CLI formats. Assisted-by: Claude Fable 5.1 via Pi --- CHANGELOG.md | 3 + CLAUDE.md | 10 +- README.md | 232 +++++++++++++++++++++++++++++++++++++++++++- skills/qmd/SKILL.md | 22 +++++ 4 files changed, 264 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2524cf779..fed98738a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,9 +6,12 @@ - Added Oxlint lint fence. - Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter `, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator`: `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, `contains`/`prefix`/`suffix` text matching, `type` for the stored type of a key's values, and `exists` presence, with an optional `caseInsensitive` flag on any condition whose value is a string (ASCII folding). Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes). +- Metadata discovery. Filtering is only useful when the caller knows what to filter on, so every surface now reports the metadata keys, types, and value counts already in the index. The CLI adds `qmd collection metadata [name...]` with `--filter ` to count only documents matching a filter (same AST as search) and `--match ` to select which metadata entries are reported, using the same AST evaluated per entry (a condition's `field` is the entry's `key` or `value`, so one grammar covers a single key, a family of keys, reverse lookup of a value, typed thresholds, type selection, and any `and`/`or`/`not` composition of those), and two windows that page: `--key-limit ` (default 50), `--key-offset `, and `--all-keys` over the keys, `--value-limit ` (default 10), `--value-offset `, and `--all-values` over the values of each key and type, plus `--sort count|value` and `--min-count ` to order and trim values. With `--filter`, the output opens with a `filter:` line and every coverage count is measured against the documents the filter admits. `qmd collection list` names each collection's top keys, `qmd collection show` details them with a value preview, and `qmd status` summarizes coverage. The SDK adds `listMetadata(options)` (the same options as `keyLimit`, `keyOffset`, `valueLimit`, `valueOffset`, `sort`, `minCount`, with `MetadataOptionError` for a value outside its domain), the MCP server adds a `metadata` tool and lists each collection's most covered key names and types in `status`, and the HTTP server adds `POST /metadata`. `MetadataFilter` and `MetadataMatch` are both `MetadataPredicate`, the one recursive grammar over the conditions a document or a metadata entry admits. Counts are documents, not values. Both windows are applied in SQL and always report the exact remainder (`totalKeys`/`remainingKeys`, `distinctValues`/`remainingValues`). Every result is read from one database snapshot. Key selection and value-count aggregation are reused within a report. Collection overviews are batched, and MCP initialization does not compute them. A string value prints bare only when the bare form is unambiguous in either layout, otherwise as an escaped JSON string. Type-conflict drill-down hints preserve metadata keys as literal shell arguments. `MetadataBindingBudgetError` rejects a filter and match that together bind more SQL parameters than the statement budget. Numbers report min, median, and max. Keys whose documents disagree on type (metadata is validated per document, so this can happen within one collection as well as across collections) are split per type with their own document counts rather than resolved. Discovery applies the same extraction gate and scope as filtering, so every value it reports is one an `eq` filter can match. Reads the existing metadata tables, so no re-indexing is needed. ### Fixed +- Filtered vector search binds candidate IDs as one JSON list, so a parser-valid metadata filter cannot exhaust Node's SQL variable limit during document lookup. Applies to both exact scans and the capped global fallback. +- Metadata discovery avoids SQLite 3.44-only aggregate ordering syntax while preserving UTF-8 order for contributing collection names. - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This keeps chunk boundaries aligned with the model that creates and verifies the diff --git a/CLAUDE.md b/CLAUDE.md index ee952e4a0..429fd56e4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -9,6 +9,7 @@ qmd collection add . --name # Create/index collection qmd collection list # List all collections with details qmd collection remove # Remove a collection by name qmd collection rename # Rename a collection +qmd collection metadata [name...] # Discover metadata keys, types, and value counts (--match, --filter, --key-limit, --value-limit) qmd init # Create a project-local .qmd index qmd ls [collection[/path]] # List collections or files in a collection qmd context add [path] "text" # Add context for path (defaults to current dir) @@ -47,9 +48,16 @@ qmd collection remove mynotes # Rename a collection qmd collection rename mynotes my-notes -# Show collection details +# Show collection details, including the top metadata keys qmd collection show mynotes +# Discover metadata keys and values to filter on +qmd collection metadata mynotes +qmd collection metadata mynotes --match '{"field":"key","operator":"eq","value":"topics"}' +qmd collection metadata mynotes --match '{"field":"value","operator":"eq","value":"docs-team"}' +qmd collection metadata mynotes --match '{"field":"key","operator":"eq","value":"topics"}' --filter '{"field":"status","operator":"eq","value":"published"}' +qmd collection metadata mynotes --key-limit 50 --key-offset 50 # next page of keys + # Set or clear the pre-update hook (runs before re-indexing on `qmd update`) qmd collection update-cmd mynotes 'git pull --ff-only' qmd collection update-cmd mynotes # clear diff --git a/README.md b/README.md index 677dbfe58..5809a88c1 100644 --- a/README.md +++ b/README.md @@ -152,6 +152,7 @@ runs in a container and a liveness probe connects from a non-loopback address. The HTTP server exposes two endpoints: - `POST /mcp` — MCP Streamable HTTP (JSON responses, stateless) - `POST /query` (alias `/search`) — structured search without the MCP protocol. Accepts the same optional `filter` object as the `query` tool (invalid filters return `400`); see [Metadata Filtering](#metadata-filtering) +- `POST /metadata` — metadata discovery without the MCP protocol. Same body as the `metadata` tool (invalid filters return `400`); see [Metadata Discovery](#metadata-discovery) - `GET /health` — liveness check with uptime @@ -198,6 +199,15 @@ Point any MCP client at `http://localhost:8181/mcp` to connect. | `get` | `maxLines` | number | Limit returned lines | | `get` | `lineNumbers` | boolean | Prefix lines with numbers (default **true**) | | `multi_get` | `pattern` | string | Glob pattern or comma-separated list | +| `metadata` | `collections` | string[] | Restrict discovery to collection names (default: the collections `query` searches) | +| `metadata` | `match` | object | Report only metadata entries matching this condition (same AST as `filter`, with `field` naming the entry's `key` or `value`) | +| `metadata` | `filter` | object | Count only documents matching this filter (same AST as `query`) | +| `metadata` | `keyLimit` | number | Keys reported (default 50). `totalKeys` and `remainingKeys` describe the rest | +| `metadata` | `keyOffset` | number | Keys skipped before the window, in report order (default 0) | +| `metadata` | `valueLimit` | number | Values reported per key and type (default 10). `remainingValues` reports the rest | +| `metadata` | `valueOffset` | number | Values skipped per key and type before the window, in `sort` order (default 0) | +| `metadata` | `sort` | string | `count` (default) or `value` | +| `metadata` | `minCount` | number | Hide values held by fewer documents (default 1) | | `multi_get` | `maxBytes` | number | Skip files larger than N (default 10240) | | `multi_get` | `maxLines` | number | Limit lines per file | | `multi_get` | `lineNumbers` | boolean | Prefix lines with numbers (default **true**) | @@ -644,9 +654,14 @@ qmd collection rename myproject my-project qmd ls notes qmd ls notes/subfolder -# Show collection details (path, glob mask, include status, context count) +# Show collection details (path, glob mask, include status, context count, top metadata keys) qmd collection show notes +# Discover metadata keys, types, and value counts (see Metadata Discovery) +qmd collection metadata notes +qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' +qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' + # Include or exclude a collection from default (unscoped) queries qmd collection include notes qmd collection exclude notes @@ -978,7 +993,7 @@ Any condition whose value is a string or an array of strings may add `"caseInsen Semantics: - Matching is typed and exact — no string/number/boolean coercion, and a type mismatch never matches (including `ne` and `nin`). Text operators match string values only. `type` matches the stored type of a key's values, which is how a filter reaches one side of a key whose documents disagree on type. -- Array-valued metadata is a set: a condition matches when any element satisfies it, `all` requires every filter value to be present. +- Array-valued metadata is a set: a condition matches when any element satisfies it, `all` requires every filter value to be present, and `ne`/`nin` require that no element equals the operand (with at least one element of the operand's type present). - Missing keys do not match `ne`/`nin`; combine with `{ "operator": "exists", "value": false }` in an `or` group to include them. - Matching is case-sensitive unless a condition sets `caseInsensitive`, which folds ASCII letters on both sides. Non-ASCII letters compare exactly. - Multiple conditions require an explicit `and` group — there is no implicit AND, and no `$`-prefixed shorthand. @@ -991,6 +1006,219 @@ Guarantees and limits: JSON output (`--format json`), the SDK, MCP structured results, and the HTTP endpoints include each result's indexed metadata. +### Metadata Discovery + +Filtering is only useful if you know what to filter on. Discovery reports the metadata keys, types, and value counts already in the index, turning "what dimensions exist" into a well-shaped filter in a few steps. It reads the same tables filtering reads: no re-indexing, and every value it reports is one an `eq` filter can match. + +Same metadata, different unit. `--filter` narrows **documents** by their metadata and decides which are counted. `--match` narrows **the metadata itself** and decides which entries are reported. Show metadata matching X from documents filtered by Y. Both take the predicate AST above, applied to a different record. A condition tests one `field` of the record under evaluation: in a filter the record is a document and `field` names one of its metadata keys, in a match the record is a metadata entry and `field` is `"key"` (the entry's key name) or `"value"` (its value). Every operator from the filter language applies, including `type`, the text operators, `caseInsensitive`, and `and`/`or`/`not` composition. Only `exists` and `all` are rejected, as they have no meaning for a single entry. + +| `--match` | Question answered | +|-----------|-------------------| +| | Which keys exist, with a window of values each | +| `{"field":"key","operator":"eq","value":"topics"}` | Everything about one key | +| `{"field":"key","operator":"prefix","value":"mem-"}` | A family of keys | +| `{"field":"key","operator":"in","value":["tags","topics","labels"]}` | Which of these key names exist | +| `{"field":"value","operator":"eq","value":"docs-team"}` | Which keys hold this value | +| `{"field":"value","operator":"prefix","value":"2025-"}` | Which keys hold values shaped like this | +| `{"field":"value","operator":"type","value":"boolean"}` | Which keys hold booleans | +| `and` of `key eq priority` and `value gte 3` | Values of one key above a threshold | +| `and` of `key eq priority` and `value type number` | The numeric side of a key whose documents disagree on type | + +Start wide and narrow: + +```sh +# Which keys does this collection use? (also shown by `qmd collection show notes`) +qmd collection metadata notes + +# Everything about one key: coverage, distinct count, top values +qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' + +# Reverse lookup: which keys hold this value +qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' + +# Values of one key matching a pattern +qmd collection metadata notes --match '{ + "operator": "and", + "operands": [ + { "field": "key", "operator": "eq", "value": "topics" }, + { "field": "value", "operator": "prefix", "value": "sql" } + ] +}' + +# What remains after a filter, before committing to it in a query +qmd collection metadata notes \ + --match '{"field":"key","operator":"eq","value":"topics"}' \ + --filter '{"field":"status","operator":"eq","value":"published"}' + +# Then search with the filter you just validated +qmd query "dependency injection" -c notes --filter '{ + "operator": "and", + "operands": [ + { "field": "status", "operator": "eq", "value": "published" }, + { "field": "topics", "operator": "all", "value": ["typescript"] } + ] +}' +``` + +The drill-down prints one block per key, in coverage order: + +```sh +qmd collection metadata notes +``` + +``` +topics string[] 388 of 480 documents 1,204 distinct + typescript 140 + sqlite 92 + search 77 + architecture 61 + sqlite-vec 44 + mcp 39 + embeddings 35 + agents 31 + cli 28 + testing 26 +1,194 more values, use --value-limit , --value-offset , or --all-values + +priority number 205 of 480 documents 5 distinct + min 1 median 3 max 5 + 1 (12) 2 (40) 3 (88) 4 (50) 5 (15) + +reviewed boolean 480 of 480 documents + true 61 false 419 +``` + +A reverse lookup answers "where does this value live" by returning every key that holds it, here a scalar key and an array key: + +```sh +qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' +``` + +``` +owner string 212 of 480 documents 1 distinct + docs-team 212 + +reviewers string[] 97 of 480 documents 1 distinct + docs-team 97 +``` + +A match on the value field selects values, not documents, so a numeric condition narrows the range it reports along with the values. Here a document holding `[1, 5]` would contribute only the `5`: + +```sh +qmd collection metadata notes --match '{ + "operator": "and", + "operands": [ + { "field": "key", "operator": "eq", "value": "priority" }, + { "field": "value", "operator": "gte", "value": 3 } + ] +}' +``` + +``` +priority number 153 of 480 documents 3 distinct + min 3 median 3 max 5 + 3 (88) 4 (50) 5 (15) +``` + +With `--filter`, the output opens with a `filter:` line stating how many documents pass, and every key's coverage is measured against that population rather than the whole collection, so the numbers you read are the numbers a filtered search would see: + +```sh +qmd collection metadata notes \ + --match '{"field":"key","operator":"eq","value":"topics"}' \ + --filter '{"field":"status","operator":"eq","value":"published"}' \ + --value-limit 5 +``` + +``` +filter: 312 of 480 documents + +topics string[] 260 of 312 documents 811 distinct + typescript 104 + sqlite 70 + search 58 + architecture 40 + mcp 31 +806 more values, use --value-limit , --value-offset , or --all-values +``` + +Two windows bound the key and value lists, and both page. The key window decides which keys are aggregated in detail and the value window decides how many values come back per key and type. Exact counts, medians, and ordering still process the relevant rows. The windows do not bound query work or the contributing collection lists. `--key-limit ` (default 50) with `--key-offset ` windows the keys, in coverage order, and `--all-keys` removes the window. `--value-limit ` (default 10) with `--value-offset ` windows the values of each key and type, in `--sort` order, and `--all-values` removes the window. `--sort count|value` orders values (count descending by default, value ascending for ranges and dates), and `--min-count ` drops the long tail. Omitting the collection name covers the default collections, exactly as an unscoped search does. + +Rules that matter when reading the output: + +- **Counts are documents, not values.** A document with `topics: [a, b]` contributes one to each. Coverage is "documents declaring this key", out of the documents the filter admits when there is one. +- **Truncation is never silent.** Every windowed list ends with the exact remainder and the flags that reach it. Structured results carry `totalKeys` and `remainingKeys` for the key window and `distinctValues` and `remainingValues` per type for the value window. +- **A bare string is exactly the value.** A string prints as-is only when nothing else could be read from it. A value that is empty, padded, contains a quote, a backslash, a control character, a line separator, or a comma or parenthesis (the delimiters of the compact `value (count)` list), or that reads as a number, boolean, or null (`"42"`, `"true"`), prints as a JSON string with every control character escaped, in both layouts, so `"a (1), b" (1)` is one value and never two. +- **One result, one snapshot.** Every count in a result is read from the same database state, even while another process is indexing. +- **Numbers report min, median, and max**, plus the enumerated values when they fit, which is enough to write a sound `gt`/`lt` threshold in one call. +- **Discovery sees exactly what filtering sees.** Same extraction gate, same active-document rule, same collection scope. Documents still pending extraction are reported on stderr and excluded until `qmd update` runs. +- **Type conflicts are reported, not resolved.** Metadata is validated one document at a time. Nothing requires two documents to agree on a key's type, whether they sit in the same collection or in different ones, so `priority: 3` in one file and `priority: high` in another both index. Discovery splits such a key by type and gives each type its own document count, which tells you how much of the corpus a typed filter would reach. Within one collection: + +```sh +qmd collection metadata work --match '{"field":"key","operator":"eq","value":"priority"}' +``` + +``` +priority number | string 1,222 of 1,620 documents + number 18 docs min 1 median 2 max 3 + string 1,204 docs high (700), medium (380), low (124) +``` + +Across collections, each type also names where it comes from: + +```sh +qmd collection metadata --match '{"field":"key","operator":"eq","value":"priority"}' +``` + +``` +priority number | string 1,427 of 2,100 documents + number 223 docs min 1 median 3 max 5 notes, work + string 1,204 docs high (700), medium (380), low (124) work +``` + +To report one side only, add a `type` condition on the value field. The filter that reaches exactly those documents is the same condition with the metadata key in `field`: + +```sh +qmd collection metadata work --match '{ + "operator": "and", + "operands": [ + { "field": "key", "operator": "eq", "value": "priority" }, + { "field": "value", "operator": "type", "value": "number" } + ] +}' +qmd query "release checklist" -c work --filter '{"field":"priority","operator":"type","value":"number"}' +``` + +`qmd collection list` names each collection's top keys, `qmd collection show ` details the top five with a value preview, and `qmd status` summarizes how many keys and files carry metadata. These summaries read current index data, so their cost depends on the metadata in scope. MCP initialization does not compute unused key summaries. + +The CLI prints text and rejects unsupported format flags. Structured discovery is available through the SDK, MCP, and HTTP with the same options (`collection`, `match`, `filter`, `keyLimit`, `keyOffset`, `valueLimit`, `valueOffset`, `sort`, `minCount`) and the same result shape: + +```typescript +// SDK: one flat result, keys in coverage order, each split per type +const discovery = await store.listMetadata({ + collection: "notes", + match: { field: "key", operator: "eq", value: "topics" }, + valueLimit: 5, +}) +discovery.documents // active documents in scope +discovery.totalKeys // keys with a matching entry, before the key window +discovery.remainingKeys // keys after the window +discovery.keys[0].types[0] // { type, multiValued, documents, distinctValues, values, remainingValues, range?, collections } + +// Narrowed by a filter: how many documents pass, and what is left to filter on +const published = await store.listMetadata({ + collection: "notes", + filter: { field: "status", operator: "eq", value: "published" }, +}) +published.filteredDocuments // the denominator for every coverage count in this result + +// Page through a wide vocabulary +const nextPage = await store.listMetadata({ keyLimit: 50, keyOffset: 50 }) +``` + +`filter` is a `MetadataFilter` and `match` is a `MetadataMatch`. Both are `MetadataPredicate`, the one recursive grammar, over the conditions each record admits, so a document-only condition (`exists`, `all`, or a metadata key as the `field`) is a type error in a match as well as a runtime one. Options outside their domain throw `MetadataOptionError`. A `filter` and `match` that each pass the grammar's limits but together bind more SQL parameters than `METADATA_SQL_BINDING_BUDGET` throw `MetadataBindingBudgetError` before any statement runs. + +The MCP `metadata` tool takes the same options with `collections` spelled as on `query`, returns the CLI shape as text and the result as `structuredContent`, and the MCP `status` tool lists each collection's most covered key names and types (with a count of the rest) so an agent's first call reveals that metadata exists. `POST /metadata` accepts the same body as the tool and returns the same result (`400` on an invalid match, filter, or option, or a filter and match over the binding budget). + ### Output Format Default output is colorized CLI format (respects `NO_COLOR` env). diff --git a/skills/qmd/SKILL.md b/skills/qmd/SKILL.md index 08b1a764a..1150353af 100644 --- a/skills/qmd/SKILL.md +++ b/skills/qmd/SKILL.md @@ -210,6 +210,28 @@ qmd query "dependency injection" --filter '{"operator":"and","operands":[{"field Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` takes one `operand`, and conditions take `field` + `value` with operators `eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), `contains`/`prefix`/`suffix` (text), `type` (the value is a `string`, `number`, or `boolean`), or `exists` (presence). Matching is typed and exact; missing keys do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to include them). Conditions with a string value may add `"caseInsensitive": true`, which folds ASCII letters. The MCP `query` tool accepts the same AST as a `filter` object. JSON output includes each result's `metadata`. +## Discover metadata before filtering + +Do not guess keys or values. `qmd collection show ` lists the top keys with types and a value preview, and `qmd collection metadata` drills in. Same metadata, different unit: `--filter` narrows documents (which are counted), `--match` narrows the metadata itself (which entries are reported). Both take the same predicate AST as `--filter` above, applied to a different record: in a match, a condition's `field` is `"key"` (the entry's key name) or `"value"` (its value), and every operator applies except `exists` and `all`: + +```bash +qmd collection metadata notes # the 50 most covered keys, ten values each +qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' # one key: coverage, distinct count, top values +qmd collection metadata notes --match '{"field":"key","operator":"prefix","value":"mem-"}' # a family of keys +qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' # reverse lookup: which keys hold this value +qmd collection metadata notes --match '{"field":"value","operator":"type","value":"boolean"}' # which keys hold booleans +qmd collection metadata notes --match '{"operator":"and","operands":[{"field":"key","operator":"eq","value":"priority"},{"field":"value","operator":"gte","value":3}]}' +qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' --filter '{"field":"status","operator":"eq","value":"published"}' +qmd collection metadata notes --key-limit 50 --key-offset 50 # the next page of keys +qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' --value-offset 10 # the next page of one key's values +``` + +Compose with `and`/`or`/`not` to ask several things in one call, and add `"caseInsensitive": true` to a string condition when the corpus mixes case. + +Read the header first: `topics string[] 388 of 480 documents 1,204 distinct` tells you coverage and cardinality before you commit to a filter. With `--filter`, a `filter: 312 of 480 documents` line opens the output and every coverage count is out of those 312. Counts are documents, not values. A value printed in double quotes is a string whose bare form would mislead (`"42"` is a string, `42` a number, `""` is empty), so paste it into a filter as the JSON string it is. A `N more values` or `N more keys` footer means a list was windowed. Page with `--value-offset`/`--key-offset` or raise `--value-limit`/`--key-limit` rather than assuming the rest. Numbers print `min`, `median`, and `max` so you can write a `gt`/`lt` threshold in one call. A key shown as `number | string` means documents disagree on type. Metadata is validated per document, so this happens within a single collection as readily as across collections. Each type reports its own document count. Filter by the type that covers the documents you want. Every value shown can be matched with `eq` under the same collection scope. + +Over MCP, call the `metadata` tool (same options, `collections` as an array) and read `totalKeys`, `remainingKeys`, `remainingValues`, and `range` from the structured result. Page keys with `keyOffset` and values with `valueOffset`. The `status` tool lists each collection's most covered key names and types, so check it first. + ## MCP Tool: `query` When using the MCP server, prefer structured searches: From 3c2a2f8e7dd345daa0b04f7dd223c15070701b35 Mon Sep 17 00:00:00 2001 From: Rik van Riel Date: Wed, 16 Sep 2026 14:33:48 -0400 Subject: [PATCH 22/82] fix(search,embed,index): stream large queries with iterate() Large result sets previously used all() which materialized full arrays in V8 heap. On an index with 13k files and 89k chunk vectors this means tens of thousands of rows (active paths, sync-state entries, pending-embedding docs, candidate chunk vectors) in a single array, which caused OOM. Now they stream via iterate(): sync-state reads, pending-embedding scans, active-path listings, glob/fuzzy scans, hash_seq eligibility (with early exit past the cap, falling back to ANN), and embedding-body fetches (substr-bound, with a bytes-only path). Small queries remain with all() as they are bounded. (cherry picked from commit 3c37c68cc8f481c92144a5a590785a69ed2b72a7) --- src/store.ts | 110 +++++++++++++++++++++++++++++++++++---------------- 1 file changed, 77 insertions(+), 33 deletions(-) diff --git a/src/store.ts b/src/store.ts index c724f0f11..5e4b716bf 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1642,11 +1642,14 @@ type FileSyncStateRow = { function getFileSyncStateMap(db: Database, collectionName: string): Map { try { - const rows = db.prepare( + const stmt = db.prepare( `SELECT relative_path, mtime_ms, size, content_hash, document_id FROM file_sync_state WHERE collection = ?` - ).all(collectionName) as FileSyncStateRow[]; + ); const map = new Map(); - for (const r of rows) map.set(r.relative_path, r); + // Large-result query: use iterate() to stream rows instead of .all() materializing at once + for (const r of stmt.iterate(collectionName) as IterableIterator) { + map.set(r.relative_path, r); + } return map; } catch { // Table may not exist yet on legacy DBs — initializeDatabase creates it on next open, @@ -2061,7 +2064,15 @@ function getPendingEmbeddingDocs(db: Database, collection?: string, model: strin GROUP BY d.hash ORDER BY MIN(d.path) `); - return (collection ? stmt.all(model, fingerprint, collection) : stmt.all(model, fingerprint)) as PendingEmbeddingDoc[]; + // Large-result query (up to 9k docs): stream via iterate() instead of .all() to bound V8 heap + const results: PendingEmbeddingDoc[] = []; + const iter = collection + ? stmt.iterate(model, fingerprint, collection) + : stmt.iterate(model, fingerprint); + for (const row of iter as IterableIterator) { + results.push(row); + } + return results; }); } @@ -2100,12 +2111,17 @@ function getEmbeddingDocsForBatch(db: Database, batch: PendingEmbeddingDoc[]): E if (batch.length === 0) return []; const placeholders = batch.map(() => "?").join(","); - const rows = db.prepare(` + // Bounded: batch size max 64 (maxDocsPerBatch), so IN list max 64 hashes. + // Use iterate() to stream rows instead of materializing all at once, nicer for large batches. + const stmt = db.prepare(` SELECT hash, doc as body FROM content WHERE hash IN (${placeholders}) - `).all(...batch.map(doc => doc.hash)) as { hash: string; body: string }[]; - const bodyByHash = new Map(rows.map(row => [row.hash, row.body])); + `); + const bodyByHash = new Map(); + for (const row of stmt.iterate(...batch.map(doc => doc.hash)) as IterableIterator<{ hash: string; body: string }>) { + bodyByHash.set(row.hash, row.body); + } return batch.map((doc) => ({ ...doc, @@ -3345,10 +3361,15 @@ export function deactivateDocument(db: Database, collectionName: string, path: s * Get all active document paths for a collection. */ export function getActiveDocumentPaths(db: Database, collectionName: string): string[] { - const rows = db.prepare(` + const stmt = db.prepare(` SELECT path FROM documents WHERE collection = ? AND active = 1 - `).all(collectionName) as { path: string }[]; - return rows.map(r => r.path); + `); + // Large-result query (up to 5k per collection): use iterate() to bound heap + const paths: string[] = []; + for (const r of stmt.iterate(collectionName) as IterableIterator<{ path: string }>) { + paths.push(r.path); + } + return paths; } export { formatQueryForEmbedding, formatDocForEmbedding }; @@ -3658,22 +3679,26 @@ export function findDocumentByDocid(db: Database, docid: string): { filepath: st } export function findSimilarFiles(db: Database, query: string, maxDistance: number = 3, limit: number = 5): string[] { - const allFiles = db.prepare(` + const stmt = db.prepare(` SELECT d.path FROM documents d WHERE d.active = 1 - `).all() as { path: string }[]; + `); + // Large-result query (all active docs, up to 9k): iterate to bound heap const queryLower = query.toLowerCase(); - const scored = allFiles - .map(f => ({ path: f.path, dist: levenshtein(f.path.toLowerCase(), queryLower) })) - .filter(f => f.dist <= maxDistance) + const scored: { path: string; dist: number }[] = []; + for (const f of stmt.iterate() as IterableIterator<{ path: string }>) { + const dist = levenshtein(f.path.toLowerCase(), queryLower); + if (dist <= maxDistance) scored.push({ path: f.path, dist }); + } + return scored .sort((a, b) => a.dist - b.dist) - .slice(0, limit); - return scored.map(f => f.path); + .slice(0, limit) + .map(f => f.path); } export function matchFilesByGlob(db: Database, pattern: string): { filepath: string; displayPath: string; bodyLength: number }[] { - const allFiles = db.prepare(` + const stmt = db.prepare(` SELECT 'qmd://' || d.collection || '/' || d.path as virtual_path, LENGTH(content.doc) as body_length, @@ -3682,16 +3707,20 @@ export function matchFilesByGlob(db: Database, pattern: string): { filepath: str FROM documents d JOIN content ON content.hash = d.hash WHERE d.active = 1 - `).all() as { virtual_path: string; body_length: number; path: string; collection: string }[]; - + `); + // Large-result query: iterate to bound heap (all active docs) const isMatch = picomatch(pattern); - return allFiles - .filter(f => isMatch(f.virtual_path) || isMatch(f.path) || isMatch(f.collection + '/' + f.path)) - .map(f => ({ - filepath: f.virtual_path, // Virtual path for precise lookup - displayPath: f.path, // Relative path for display - bodyLength: f.body_length - })); + const results: { filepath: string; displayPath: string; bodyLength: number }[] = []; + for (const f of stmt.iterate() as IterableIterator<{ virtual_path: string; body_length: number; path: string; collection: string }>) { + if (isMatch(f.virtual_path) || isMatch(f.path) || isMatch(f.collection + '/' + f.path)) { + results.push({ + filepath: f.virtual_path, + displayPath: f.path, + bodyLength: f.body_length, + }); + } + } + return results; } // ============================================================================= @@ -4497,9 +4526,16 @@ export async function searchVec(db: Database, query: string, model: string, limi eligibleSql += ` WHERE ${eligibleConditions.join(" AND ")}`; - const eligibleHashSeqs = withLazyContentVectorMigration(db, () => - db.prepare(eligibleSql).all(...eligibleParams) as { hash_seq: string }[], - ).map((r) => r.hash_seq); + const eligibleHashSeqs: string[] = withLazyContentVectorMigration(db, () => { + const stmt = db.prepare(eligibleSql); + // Large-result query (up to 20k): use iterate() to bound heap, early exit if over max + const seqs: string[] = []; + for (const r of stmt.iterate(...eligibleParams) as IterableIterator<{ hash_seq: string }>) { + seqs.push(r.hash_seq); + if (seqs.length > FILTERED_VEC_EXACT_SCAN_MAX) break; + } + return seqs; + }); if (eligibleHashSeqs.length === 0) return []; @@ -4611,8 +4647,9 @@ async function getEmbedding(text: string, model: string, isQuery: boolean, sessi */ export function getHashesForEmbedding(db: Database, model: string = DEFAULT_EMBED_MODEL): { hash: string; body: string; path: string }[] { const fingerprint = getEmbeddingFingerprint(model); - return withLazyContentVectorMigration(db, () => db.prepare(` - SELECT d.hash, c.doc as body, MIN(d.path) as path + return withLazyContentVectorMigration(db, () => { + const stmt = db.prepare(` + SELECT d.hash, substr(c.doc, 1, 262144) as body, MIN(d.path) as path FROM documents d JOIN content c ON d.hash = c.hash LEFT JOIN ( @@ -4624,7 +4661,14 @@ export function getHashesForEmbedding(db: Database, model: string = DEFAULT_EMBE WHERE d.active = 1 AND (v.hash IS NULL OR v.chunk_count < v.expected_chunks) GROUP BY d.hash - `).all(model, fingerprint) as { hash: string; body: string; path: string }[]); + `); + // Large-result query (up to 9k): use iterate() to stream, bound heap + const results: { hash: string; body: string; path: string }[] = []; + for (const row of stmt.iterate(model, fingerprint) as IterableIterator<{ hash: string; body: string; path: string }>) { + results.push(row); + } + return results; + }); } /** From 86966ddfbfc7b8f7cf6951b0f5ed625e08168c6d Mon Sep 17 00:00:00 2001 From: Aaron Casanova Date: Thu, 17 Sep 2026 00:18:11 -0700 Subject: [PATCH 23/82] refactor(metadata): trim discovery to what ships A holistic pass over the feature after the correctness and performance audits, asking whether each piece is as small and intentional as it can be. No behavior changes. The full matrix, the close-out packet's semantic checks (adapted for the removed helpers), and the SQLite 3.43.2 path all pass on Bun and Node. Code: - Remove `listMetadataKeys()` and `getMetadataOverview()` from metadata-store.ts. Neither had a production caller: every status view reads `listMetadataCollectionSummaries()`. Their doc comment described the pre-batching design. `queryMetadataOverviews()` loses the SQL fragment parameter that only they varied. - Stop exporting `DEFAULT_METADATA_KEY_LIMIT`, `DEFAULT_METADATA_VALUE_LIMIT`, and `METADATA_SQL_BINDING_BUDGET` from the SDK. The defaults are in the option docs and the budget is an internal safety valve. The error classes stay exported. - Un-export `formatMetadataKeySummary()`, which only its own module calls. - Measure text-operator byte lengths with `Buffer.byteLength`, the idiom the rest of the file already uses, instead of a second `TextEncoder`. - Merge the two stacked comments on `warnPendingMetadata()` into one. - Empty-result messages on the CLI and MCP tool say "see which keys exist" rather than "see every key", since the default window is 50. - Note why vitest's typecheck ignores source errors: it would otherwise report pre-existing errors under tools/, and tsc covers src/. Tests: - Fold the `listMetadataKeys` assertions into the batched-overview tests. - In the "rank once, group once" test, keep the statement-count guard and the materialization guard (the ranking CTE plans as MATERIALIZE, never CO-ROUTINE, with and without statistics). One ranking statement does not mean one evaluation of the ranking, and the guard fails the moment MATERIALIZED is dropped. Remove the EXISTS and LIST SUBQUERY assertions, which pin the filter compiler's plan and belong to its own suite. Docs, sized to sit beside their neighbors: - README: list the `metadata` tool under "Tools exposed" (it was missing) and move its parameter rows out of the middle of `multi_get`'s. Lead the Discovery section with the one-sentence model, keep one example per idea, render the window flags in the README's own flag-list style, shorten the reading rules, and drop the internals (binding budget constant, status cost, MCP initialization) that belong in the PR. - CHANGELOG: trim the discovery entry to user-visible behavior. - MCP `metadata` tool description: cut to the mental model, four match shapes, and how to read and page the result. It is now shorter than the `query` tool's. - skills/qmd/SKILL.md: restructure the two metadata sections in the skill's own style (wrapped prose, short paragraphs, bulleted reading rules, five examples instead of nine). Add one sentence of judgment in the register the skill already uses for `-c`: add a metadata filter when it expresses a constraint the user intends, and check coverage first, because a value condition only sees documents that declare its key. A matching "Do not filter blind" pitfall. Bump the skill version to 2.3.0. Assisted-by: Claude Fable 5.1 via Pi --- CHANGELOG.md | 3 +- README.md | 96 +++++++++++---------------------- skills/qmd/SKILL.md | 69 ++++++++++++++++++------ src/cli/qmd.ts | 8 +-- src/index.ts | 13 +---- src/mcp/server.ts | 22 +++----- src/metadata-filter.ts | 8 +-- src/metadata-format.ts | 2 +- src/metadata-store.ts | 29 ++++------ test/metadata-cli.test.ts | 2 +- test/metadata-discovery.test.ts | 59 ++++++++++---------- test/metadata-surfaces.test.ts | 2 +- vitest.config.ts | 1 + 13 files changed, 140 insertions(+), 174 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index fed98738a..bb094c48e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,12 +6,11 @@ - Added Oxlint lint fence. - Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter `, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator`: `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, `contains`/`prefix`/`suffix` text matching, `type` for the stored type of a key's values, and `exists` presence, with an optional `caseInsensitive` flag on any condition whose value is a string (ASCII folding). Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes). -- Metadata discovery. Filtering is only useful when the caller knows what to filter on, so every surface now reports the metadata keys, types, and value counts already in the index. The CLI adds `qmd collection metadata [name...]` with `--filter ` to count only documents matching a filter (same AST as search) and `--match ` to select which metadata entries are reported, using the same AST evaluated per entry (a condition's `field` is the entry's `key` or `value`, so one grammar covers a single key, a family of keys, reverse lookup of a value, typed thresholds, type selection, and any `and`/`or`/`not` composition of those), and two windows that page: `--key-limit ` (default 50), `--key-offset `, and `--all-keys` over the keys, `--value-limit ` (default 10), `--value-offset `, and `--all-values` over the values of each key and type, plus `--sort count|value` and `--min-count ` to order and trim values. With `--filter`, the output opens with a `filter:` line and every coverage count is measured against the documents the filter admits. `qmd collection list` names each collection's top keys, `qmd collection show` details them with a value preview, and `qmd status` summarizes coverage. The SDK adds `listMetadata(options)` (the same options as `keyLimit`, `keyOffset`, `valueLimit`, `valueOffset`, `sort`, `minCount`, with `MetadataOptionError` for a value outside its domain), the MCP server adds a `metadata` tool and lists each collection's most covered key names and types in `status`, and the HTTP server adds `POST /metadata`. `MetadataFilter` and `MetadataMatch` are both `MetadataPredicate`, the one recursive grammar over the conditions a document or a metadata entry admits. Counts are documents, not values. Both windows are applied in SQL and always report the exact remainder (`totalKeys`/`remainingKeys`, `distinctValues`/`remainingValues`). Every result is read from one database snapshot. Key selection and value-count aggregation are reused within a report. Collection overviews are batched, and MCP initialization does not compute them. A string value prints bare only when the bare form is unambiguous in either layout, otherwise as an escaped JSON string. Type-conflict drill-down hints preserve metadata keys as literal shell arguments. `MetadataBindingBudgetError` rejects a filter and match that together bind more SQL parameters than the statement budget. Numbers report min, median, and max. Keys whose documents disagree on type (metadata is validated per document, so this can happen within one collection as well as across collections) are split per type with their own document counts rather than resolved. Discovery applies the same extraction gate and scope as filtering, so every value it reports is one an `eq` filter can match. Reads the existing metadata tables, so no re-indexing is needed. +- Metadata discovery. Filtering is only useful when the caller knows what to filter on, so every surface now reports the metadata keys, types, and value counts already in the index — read from the existing metadata tables, with no re-indexing. The CLI adds `qmd collection metadata [name...]` with `--filter ` to count only documents matching a filter (same AST as search) and `--match ` to select which metadata entries are reported (the same AST evaluated per entry, where a condition's `field` is the entry's `key` or `value`, so one grammar covers a single key, a family of keys, reverse lookup of a value, typed thresholds, and any `and`/`or`/`not` composition), plus `--key-limit`/`--key-offset`/`--all-keys` and `--value-limit`/`--value-offset`/`--all-values` to page both windows, and `--sort count|value` and `--min-count ` to order and trim values. Every windowed list reports its exact remainder, counts are documents rather than values, numbers report min/median/max, keys whose documents disagree on type are split per type with their own counts, and with `--filter` every coverage count is measured against the documents the filter admits. `qmd collection list` names each collection's top keys, `qmd collection show` details them with a value preview, and `qmd status` summarizes coverage. The SDK adds `listMetadata(options)` (the same options, spelled `keyLimit`, `valueLimit`, and so on, with `MetadataOptionError` for a value outside its domain and `MetadataBindingBudgetError` for a filter and match that together bind too many SQL parameters), the MCP server adds a `metadata` tool and lists each collection's most covered keys in `status`, and the HTTP server adds `POST /metadata`. `MetadataFilter` and `MetadataMatch` are both `MetadataPredicate`, one recursive grammar over the conditions a document or a metadata entry admits. ### Fixed - Filtered vector search binds candidate IDs as one JSON list, so a parser-valid metadata filter cannot exhaust Node's SQL variable limit during document lookup. Applies to both exact scans and the capped global fallback. -- Metadata discovery avoids SQLite 3.44-only aggregate ordering syntax while preserving UTF-8 order for contributing collection names. - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This keeps chunk boundaries aligned with the model that creates and verifies the diff --git a/README.md b/README.md index 5809a88c1..d1b50c396 100644 --- a/README.md +++ b/README.md @@ -94,7 +94,8 @@ Although the tool works perfectly fine when you just tell your agent to use it o - `query` — Search with typed sub-queries (`lex`/`vec`/`hyde`), combined via RRF + reranking - `get` — Retrieve a document by path or docid (with fuzzy matching suggestions) - `multi_get` — Batch retrieve by glob pattern, comma-separated list, or docids -- `status` — Index health and collection info +- `status` — Index health and collection info, including each collection's top metadata keys +- `metadata` — Discover metadata keys, types, and value counts to filter on **Claude Desktop configuration** (`~/Library/Application Support/Claude/claude_desktop_config.json`): @@ -199,6 +200,9 @@ Point any MCP client at `http://localhost:8181/mcp` to connect. | `get` | `maxLines` | number | Limit returned lines | | `get` | `lineNumbers` | boolean | Prefix lines with numbers (default **true**) | | `multi_get` | `pattern` | string | Glob pattern or comma-separated list | +| `multi_get` | `maxBytes` | number | Skip files larger than N (default 10240) | +| `multi_get` | `maxLines` | number | Limit lines per file | +| `multi_get` | `lineNumbers` | boolean | Prefix lines with numbers (default **true**) | | `metadata` | `collections` | string[] | Restrict discovery to collection names (default: the collections `query` searches) | | `metadata` | `match` | object | Report only metadata entries matching this condition (same AST as `filter`, with `field` naming the entry's `key` or `value`) | | `metadata` | `filter` | object | Count only documents matching this filter (same AST as `query`) | @@ -208,9 +212,6 @@ Point any MCP client at `http://localhost:8181/mcp` to connect. | `metadata` | `valueOffset` | number | Values skipped per key and type before the window, in `sort` order (default 0) | | `metadata` | `sort` | string | `count` (default) or `value` | | `metadata` | `minCount` | number | Hide values held by fewer documents (default 1) | -| `multi_get` | `maxBytes` | number | Skip files larger than N (default 10240) | -| `multi_get` | `maxLines` | number | Limit lines per file | -| `multi_get` | `lineNumbers` | boolean | Prefix lines with numbers (default **true**) | Unknown parameters are silently ignored (not rejected) — double-check names if results seem unscoped. The HTTP `/query` and `/search` endpoints return @@ -1001,16 +1002,16 @@ Semantics: Guarantees and limits: - Every returned result satisfies the filter, before RRF fusion and reranking. -- Like collection filtering, highly selective filters are best-effort for top-K completeness: backends over-fetch and post-filter, so a very selective filter can return fewer than `limit` results. +- Highly selective filters can return fewer than `limit` results. Lexical search filters a bounded over-fetch window; vector search scans the filter-eligible set exactly when it holds at most 20,000 chunk vectors, and above that over-fetches and post-filters. - Filtered search only considers documents whose metadata has been extracted (run `qmd update` after upgrading; `qmd status` shows the pending count). JSON output (`--format json`), the SDK, MCP structured results, and the HTTP endpoints include each result's indexed metadata. ### Metadata Discovery -Filtering is only useful if you know what to filter on. Discovery reports the metadata keys, types, and value counts already in the index, turning "what dimensions exist" into a well-shaped filter in a few steps. It reads the same tables filtering reads: no re-indexing, and every value it reports is one an `eq` filter can match. +Filtering is only useful if you know what to filter on. Discovery reports the metadata keys, types, and value counts already in the index, so a filter can be written from what is indexed instead of guessed. It reads the same tables filtering reads: no re-indexing, and every value it reports is one an `eq` filter can match. -Same metadata, different unit. `--filter` narrows **documents** by their metadata and decides which are counted. `--match` narrows **the metadata itself** and decides which entries are reported. Show metadata matching X from documents filtered by Y. Both take the predicate AST above, applied to a different record. A condition tests one `field` of the record under evaluation: in a filter the record is a document and `field` names one of its metadata keys, in a match the record is a metadata entry and `field` is `"key"` (the entry's key name) or `"value"` (its value). Every operator from the filter language applies, including `type`, the text operators, `caseInsensitive`, and `and`/`or`/`not` composition. Only `exists` and `all` are rejected, as they have no meaning for a single entry. +The command is one sentence: show metadata matching X for documents filtered by Y. `--filter` narrows **documents** by their metadata (same AST as search) and decides which are counted. `--match` narrows **the metadata itself** and decides which entries are reported. It takes the same AST, evaluated against each metadata entry instead of each document: a condition's `field` is `"key"` (the entry's key name) or `"value"` (its value). Every operator applies, including `type`, the text operators, `caseInsensitive`, and `and`/`or`/`not`. Only `exists` and `all` are rejected, as they have no meaning for a single entry. | `--match` | Question answered | |-----------|-------------------| @@ -1027,7 +1028,7 @@ Same metadata, different unit. `--filter` narrows **documents** by their metadat Start wide and narrow: ```sh -# Which keys does this collection use? (also shown by `qmd collection show notes`) +# Which keys does this collection use? (`qmd collection show notes` previews the top five) qmd collection metadata notes # Everything about one key: coverage, distinct count, top values @@ -1088,39 +1089,7 @@ reviewed boolean 480 of 480 documents true 61 false 419 ``` -A reverse lookup answers "where does this value live" by returning every key that holds it, here a scalar key and an array key: - -```sh -qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' -``` - -``` -owner string 212 of 480 documents 1 distinct - docs-team 212 - -reviewers string[] 97 of 480 documents 1 distinct - docs-team 97 -``` - -A match on the value field selects values, not documents, so a numeric condition narrows the range it reports along with the values. Here a document holding `[1, 5]` would contribute only the `5`: - -```sh -qmd collection metadata notes --match '{ - "operator": "and", - "operands": [ - { "field": "key", "operator": "eq", "value": "priority" }, - { "field": "value", "operator": "gte", "value": 3 } - ] -}' -``` - -``` -priority number 153 of 480 documents 3 distinct - min 3 median 3 max 5 - 3 (88) 4 (50) 5 (15) -``` - -With `--filter`, the output opens with a `filter:` line stating how many documents pass, and every key's coverage is measured against that population rather than the whole collection, so the numbers you read are the numbers a filtered search would see: +With `--filter`, the output opens with a `filter:` line stating how many documents pass, and every coverage count is measured against that population rather than the whole collection — the numbers a filtered search would see: ```sh qmd collection metadata notes \ @@ -1141,29 +1110,28 @@ topics string[] 260 of 312 documents 811 distinct 806 more values, use --value-limit , --value-offset , or --all-values ``` -Two windows bound the key and value lists, and both page. The key window decides which keys are aggregated in detail and the value window decides how many values come back per key and type. Exact counts, medians, and ordering still process the relevant rows. The windows do not bound query work or the contributing collection lists. `--key-limit ` (default 50) with `--key-offset ` windows the keys, in coverage order, and `--all-keys` removes the window. `--value-limit ` (default 10) with `--value-offset ` windows the values of each key and type, in `--sort` order, and `--all-values` removes the window. `--sort count|value` orders values (count descending by default, value ascending for ranges and dates), and `--min-count ` drops the long tail. Omitting the collection name covers the default collections, exactly as an unscoped search does. - -Rules that matter when reading the output: - -- **Counts are documents, not values.** A document with `topics: [a, b]` contributes one to each. Coverage is "documents declaring this key", out of the documents the filter admits when there is one. -- **Truncation is never silent.** Every windowed list ends with the exact remainder and the flags that reach it. Structured results carry `totalKeys` and `remainingKeys` for the key window and `distinctValues` and `remainingValues` per type for the value window. -- **A bare string is exactly the value.** A string prints as-is only when nothing else could be read from it. A value that is empty, padded, contains a quote, a backslash, a control character, a line separator, or a comma or parenthesis (the delimiters of the compact `value (count)` list), or that reads as a number, boolean, or null (`"42"`, `"true"`), prints as a JSON string with every control character escaped, in both layouts, so `"a (1), b" (1)` is one value and never two. -- **One result, one snapshot.** Every count in a result is read from the same database state, even while another process is indexing. -- **Numbers report min, median, and max**, plus the enumerated values when they fit, which is enough to write a sound `gt`/`lt` threshold in one call. -- **Discovery sees exactly what filtering sees.** Same extraction gate, same active-document rule, same collection scope. Documents still pending extraction are reported on stderr and excluded until `qmd update` runs. -- **Type conflicts are reported, not resolved.** Metadata is validated one document at a time. Nothing requires two documents to agree on a key's type, whether they sit in the same collection or in different ones, so `priority: 3` in one file and `priority: high` in another both index. Discovery splits such a key by type and gives each type its own document count, which tells you how much of the corpus a typed filter would reach. Within one collection: +Two windows page the key and value lists. Each windowed list ends with the exact remainder and the flags that reach it: ```sh -qmd collection metadata work --match '{"field":"key","operator":"eq","value":"priority"}' +--key-limit # Keys reported, in coverage order (default 50) +--key-offset # Keys skipped before the window +--all-keys # Remove the key window +--value-limit # Values reported per key and type (default 10) +--value-offset # Values skipped per key and type, in --sort order +--all-values # Remove the value window +--sort count|value # Order values by document count (default) or by value +--min-count # Drop values held by fewer documents ``` -``` -priority number | string 1,222 of 1,620 documents - number 18 docs min 1 median 2 max 3 - string 1,204 docs high (700), medium (380), low (124) -``` +Omitting the collection name covers the default collections, exactly as an unscoped search does. -Across collections, each type also names where it comes from: +Reading the output: + +- **Counts are documents, not values.** A document with `topics: [a, b]` contributes one to each. Coverage is "documents declaring this key", out of the documents the filter admits when there is one. +- **A bare string is exactly the value.** A string whose bare form could be read as something else (`"42"`, `""`, `"a, b"`) prints as a JSON string, so paste it into a filter as the JSON string it is. +- **Numbers report min, median, and max**, plus the enumerated values when they fit, which is enough to write a `gt`/`lt` threshold in one call. +- **Discovery sees exactly what filtering sees.** Same extraction gate, same active-document rule, same collection scope, and every count in a result comes from one database snapshot. Documents still pending extraction are reported on stderr and excluded until `qmd update` runs. +- **Type conflicts are reported, not resolved.** Metadata is validated one document at a time, so `priority: 3` in one file and `priority: high` in another both index, within one collection or across several. Discovery splits such a key by type, each with its own document count and (across collections) its contributing collections: ```sh qmd collection metadata --match '{"field":"key","operator":"eq","value":"priority"}' @@ -1178,17 +1146,17 @@ priority number | string 1,427 of 2,100 documents To report one side only, add a `type` condition on the value field. The filter that reaches exactly those documents is the same condition with the metadata key in `field`: ```sh -qmd collection metadata work --match '{ +qmd collection metadata --match '{ "operator": "and", "operands": [ { "field": "key", "operator": "eq", "value": "priority" }, { "field": "value", "operator": "type", "value": "number" } ] }' -qmd query "release checklist" -c work --filter '{"field":"priority","operator":"type","value":"number"}' +qmd query "release checklist" --filter '{"field":"priority","operator":"type","value":"number"}' ``` -`qmd collection list` names each collection's top keys, `qmd collection show ` details the top five with a value preview, and `qmd status` summarizes how many keys and files carry metadata. These summaries read current index data, so their cost depends on the metadata in scope. MCP initialization does not compute unused key summaries. +`qmd collection list` names each collection's top keys, `qmd collection show ` details the top five with a value preview, and `qmd status` summarizes how many keys and files carry metadata. The CLI prints text and rejects unsupported format flags. Structured discovery is available through the SDK, MCP, and HTTP with the same options (`collection`, `match`, `filter`, `keyLimit`, `keyOffset`, `valueLimit`, `valueOffset`, `sort`, `minCount`) and the same result shape: @@ -1215,9 +1183,9 @@ published.filteredDocuments // the denominator for every coverage count in th const nextPage = await store.listMetadata({ keyLimit: 50, keyOffset: 50 }) ``` -`filter` is a `MetadataFilter` and `match` is a `MetadataMatch`. Both are `MetadataPredicate`, the one recursive grammar, over the conditions each record admits, so a document-only condition (`exists`, `all`, or a metadata key as the `field`) is a type error in a match as well as a runtime one. Options outside their domain throw `MetadataOptionError`. A `filter` and `match` that each pass the grammar's limits but together bind more SQL parameters than `METADATA_SQL_BINDING_BUDGET` throw `MetadataBindingBudgetError` before any statement runs. +`filter` is a `MetadataFilter` and `match` is a `MetadataMatch`. Both are `MetadataPredicate`, the one recursive grammar over the conditions each record admits, so a document-only condition (`exists`, `all`, or a metadata key as the `field`) is a type error in a match as well as a runtime one. An option outside its domain throws `MetadataOptionError`, and a `filter` and `match` that together bind more SQL parameters than one statement allows throw `MetadataBindingBudgetError`. -The MCP `metadata` tool takes the same options with `collections` spelled as on `query`, returns the CLI shape as text and the result as `structuredContent`, and the MCP `status` tool lists each collection's most covered key names and types (with a count of the rest) so an agent's first call reveals that metadata exists. `POST /metadata` accepts the same body as the tool and returns the same result (`400` on an invalid match, filter, or option, or a filter and match over the binding budget). +The MCP `metadata` tool takes the same options with `collections` spelled as on `query`, returns the CLI text plus the result as `structuredContent`, and the MCP `status` tool lists each collection's most covered keys so an agent's first call reveals that metadata exists. `POST /metadata` accepts the same body as the tool and returns the same result (`400` on an invalid match, filter, or option). ### Output Format diff --git a/skills/qmd/SKILL.md b/skills/qmd/SKILL.md index 1150353af..e80a3dbbd 100644 --- a/skills/qmd/SKILL.md +++ b/skills/qmd/SKILL.md @@ -5,7 +5,7 @@ license: MIT compatibility: Requires qmd CLI or MCP server. Install via `npm install -g @tobilu/qmd`. metadata: author: tobi - version: "2.2.0" + version: "2.3.0" allowed-tools: Bash(qmd:*), mcp__qmd__* --- @@ -201,36 +201,70 @@ Omit `-c` to search everything. ## Filter by metadata -Documents can carry typed metadata in a `qmd.metadata` frontmatter block (strings, numbers, booleans, or flat arrays). `search`, `vsearch`, and `query` accept `--filter` with a recursive JSON AST; every returned result satisfies it: +Documents can carry typed metadata in a `qmd.metadata` frontmatter block +(strings, numbers, booleans, or flat arrays). `search`, `vsearch`, and `query` +accept `--filter` with a recursive JSON AST, and every returned result satisfies +it. Add a metadata filter when it expresses a constraint the user intends, not +by default, and check coverage first: a condition on a key's value only sees +documents that declare that key, and discovery (below) shows how many do. ```bash qmd search "authentication" --filter '{"field":"status","operator":"eq","value":"published"}' qmd query "dependency injection" --filter '{"operator":"and","operands":[{"field":"topics","operator":"all","value":["typescript"]},{"field":"status","operator":"nin","value":["draft","archived"]}]}' ``` -Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` takes one `operand`, and conditions take `field` + `value` with operators `eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), `contains`/`prefix`/`suffix` (text), `type` (the value is a `string`, `number`, or `boolean`), or `exists` (presence). Matching is typed and exact; missing keys do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to include them). Conditions with a string value may add `"caseInsensitive": true`, which folds ASCII letters. The MCP `query` tool accepts the same AST as a `filter` object. JSON output includes each result's `metadata`. +Nodes are discriminated by `operator`: groups `and`/`or` take `operands`, `not` +takes one `operand`, and conditions take `field` + `value` with operators +`eq`/`ne`/`gt`/`gte`/`lt`/`lte` (comparison), `in`/`nin`/`all` (membership), +`contains`/`prefix`/`suffix` (text), `type` (the value is a `string`, `number`, +or `boolean`), or `exists` (presence). Matching is typed and exact; missing keys +do not match `ne`/`nin` (add an `exists: false` branch in an `or` group to +include them). Conditions with a string value may add `"caseInsensitive": true`, +which folds ASCII letters. The MCP `query` tool accepts the same AST as a +`filter` object. JSON output includes each result's `metadata`. ## Discover metadata before filtering -Do not guess keys or values. `qmd collection show ` lists the top keys with types and a value preview, and `qmd collection metadata` drills in. Same metadata, different unit: `--filter` narrows documents (which are counted), `--match` narrows the metadata itself (which entries are reported). Both take the same predicate AST as `--filter` above, applied to a different record: in a match, a condition's `field` is `"key"` (the entry's key name) or `"value"` (its value), and every operator applies except `exists` and `all`: +Do not guess keys or values. `qmd collection show ` lists the top keys +with types and a value preview, and `qmd collection metadata` drills in. Same +metadata, different unit: `--filter` narrows documents (which are counted), +`--match` narrows the metadata itself (which entries are reported). Both take +the predicate AST above. In a match, a condition's `field` is `"key"` (the +entry's key name) or `"value"` (its value), and every operator applies except +`exists` and `all`: ```bash -qmd collection metadata notes # the 50 most covered keys, ten values each -qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' # one key: coverage, distinct count, top values -qmd collection metadata notes --match '{"field":"key","operator":"prefix","value":"mem-"}' # a family of keys -qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' # reverse lookup: which keys hold this value -qmd collection metadata notes --match '{"field":"value","operator":"type","value":"boolean"}' # which keys hold booleans +qmd collection metadata notes # top keys, top values +qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' # one key in depth +qmd collection metadata notes --match '{"field":"value","operator":"eq","value":"docs-team"}' # which keys hold this value qmd collection metadata notes --match '{"operator":"and","operands":[{"field":"key","operator":"eq","value":"priority"},{"field":"value","operator":"gte","value":3}]}' qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' --filter '{"field":"status","operator":"eq","value":"published"}' -qmd collection metadata notes --key-limit 50 --key-offset 50 # the next page of keys -qmd collection metadata notes --match '{"field":"key","operator":"eq","value":"topics"}' --value-offset 10 # the next page of one key's values ``` -Compose with `and`/`or`/`not` to ask several things in one call, and add `"caseInsensitive": true` to a string condition when the corpus mixes case. - -Read the header first: `topics string[] 388 of 480 documents 1,204 distinct` tells you coverage and cardinality before you commit to a filter. With `--filter`, a `filter: 312 of 480 documents` line opens the output and every coverage count is out of those 312. Counts are documents, not values. A value printed in double quotes is a string whose bare form would mislead (`"42"` is a string, `42` a number, `""` is empty), so paste it into a filter as the JSON string it is. A `N more values` or `N more keys` footer means a list was windowed. Page with `--value-offset`/`--key-offset` or raise `--value-limit`/`--key-limit` rather than assuming the rest. Numbers print `min`, `median`, and `max` so you can write a `gt`/`lt` threshold in one call. A key shown as `number | string` means documents disagree on type. Metadata is validated per document, so this happens within a single collection as readily as across collections. Each type reports its own document count. Filter by the type that covers the documents you want. Every value shown can be matched with `eq` under the same collection scope. - -Over MCP, call the `metadata` tool (same options, `collections` as an array) and read `totalKeys`, `remainingKeys`, `remainingValues`, and `range` from the structured result. Page keys with `keyOffset` and values with `valueOffset`. The `status` tool lists each collection's most covered key names and types, so check it first. +The last form shows what a filter leaves behind before you commit to it in a +query. Read the output like this: + +- **The header is the decision.** `topics string[] 388 of 480 documents + 1,204 distinct` gives coverage and cardinality. With `--filter`, a + `filter: 312 of 480 documents` line opens the output and every count is out + of those 312. Counts are documents, not values. +- **Quotes mark a string that reads like something else.** `"42"` is a string, + `42` a number, `""` is empty. Paste a quoted value into a filter as the JSON + string it is. +- **Footers mean windowed.** `N more values` or `N more keys` means the list + was cut. Page with `--value-offset`/`--key-offset` or raise + `--value-limit`/`--key-limit` rather than assuming the rest. +- **Numbers print `min`, `median`, `max`**, enough to write a `gt`/`lt` + threshold in one call. +- **`number | string` means documents disagree on type.** Each type reports its + own document count. Filter by the type that covers the documents you want. + +Every value shown can be matched with `eq` under the same collection scope. + +Over MCP, call the `metadata` tool (same options, `collections` as an array) +and read `totalKeys`, `remainingKeys`, `remainingValues`, and `range` from the +structured result. Page keys with `keyOffset` and values with `valueOffset`. +The `status` tool lists each collection's most covered keys, so check it first. ## MCP Tool: `query` @@ -316,6 +350,9 @@ server configuration. have. You expand the query; the model just ranks. - **Do not overuse semantic search.** If you know exact titles or terms, BM25 is faster and often better. +- **Do not filter blind.** Run `qmd collection metadata` first and read the + coverage. A key declared by 388 of 480 documents leaves 92 that no value + condition on it can reach; only `exists: false` selects them. - **Do not mutate indexes casually.** `qmd collection add`, `qmd update`, and `qmd embed` change local state and can be expensive. - **Model-backed commands can be environment-sensitive.** If `qmd query`, diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 946385643..318638d87 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -1992,7 +1992,7 @@ function collectionMetadata(collectionNames: string[], options: ListMetadataOpti keyOffset: options.keyOffset ?? 0, keyOffsetLabel: "--key-offset", emptyMessage: selection - ? "No metadata matches. Run 'qmd collection metadata' without --match or --filter to see every key." + ? "No metadata matches. Run 'qmd collection metadata' without --match or --filter to see which keys exist." : "No metadata found. Add qmd.metadata frontmatter and run 'qmd update'.", colors: c, })); @@ -3032,9 +3032,9 @@ function parseCliPredicateFlag(raw: unknown, predicateFlag: CliPredic } } -// Filtered search excludes documents without current metadata extraction; -// tell the user when that makes results incomplete. -/** Warn about documents the extraction gate excludes, within the collections the command reads. */ +// Filtered search and discovery exclude documents without current metadata +// extraction; tell the user when that makes results incomplete, scoped to +// the collections the command reads. function warnPendingMetadata(db: Database, collectionNames: string[]): void { const pendingMetadata = countDocumentsPendingMetadata(db, collectionNames.length > 0 ? collectionNames : undefined); if (pendingMetadata === 0) return; diff --git a/src/index.ts b/src/index.ts index 7bb50d9f2..1e7fefcb0 100644 --- a/src/index.ts +++ b/src/index.ts @@ -94,9 +94,6 @@ import { listMetadata as storeListMetadata, MetadataBindingBudgetError, MetadataOptionError, - DEFAULT_METADATA_KEY_LIMIT, - DEFAULT_METADATA_VALUE_LIMIT, - METADATA_SQL_BINDING_BUDGET, type ListMetadataOptions, type ListMetadataResult, type MetadataKeySummary, @@ -174,13 +171,7 @@ export type { MetadataValueCount, MetadataKeyOverview, }; -export { - MetadataBindingBudgetError, - MetadataOptionError, - DEFAULT_METADATA_KEY_LIMIT, - DEFAULT_METADATA_VALUE_LIMIT, - METADATA_SQL_BINDING_BUDGET, -}; +export { MetadataBindingBudgetError, MetadataOptionError }; // Re-export the internal Store type for advanced consumers export type { InternalStore }; @@ -361,7 +352,7 @@ export interface QMDStore { * snapshot. Throws MetadataFilterError for an invalid predicate, * MetadataOptionError for an option outside its domain, and * MetadataBindingBudgetError when `filter` and `match` together bind more - * SQL parameters than METADATA_SQL_BINDING_BUDGET. + * SQL parameters than one statement allows. */ listMetadata(options?: ListMetadataOptions): Promise; diff --git a/src/mcp/server.ts b/src/mcp/server.ts index 171622515..41b3a0e75 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -669,35 +669,25 @@ Intent-aware lex (C++ performance, not sports): "metadata", { title: "Metadata Discovery", - description: `Discover which metadata keys exist, what types they hold, and how many documents share each value, so you can write a precise \`filter\` for the query tool. - -Documents carry metadata as \`qmd.metadata\` frontmatter (strings, numbers, booleans, or arrays of one of those). This tool reports what is indexed, never guesses. + description: `Discover which metadata keys exist, what types they hold, and how many documents share each value, so you can write a \`filter\` for the query tool from what is indexed instead of guessing. ## Mental model -\`filter\` selects WHICH documents are counted. \`match\` selects WHICH metadata entries of those documents are reported. Both take the same recursive AST as the query tool's \`filter\`. A condition tests one \`field\` of the record under evaluation: for \`filter\` the record is a document and \`field\` names one of its metadata keys, for \`match\` the record is a metadata entry and \`field\` is \`"key"\` (the entry's key name) or \`"value"\` (its value). Every operator applies (eq/ne/gt/gte/lt/lte, in/nin, contains/prefix/suffix, type, and/or/not, caseInsensitive), except \`exists\` and \`all\`, which have no meaning for a single entry. +\`filter\` selects WHICH documents are counted. \`match\` selects WHICH metadata entries of those documents are reported. Both take the same recursive AST as the query tool's \`filter\`. A condition tests one \`field\` of the record under evaluation: for \`filter\` the record is a document and \`field\` names one of its metadata keys, for \`match\` the record is a metadata entry and \`field\` is \`"key"\` or \`"value"\`. Every operator applies to a match except \`exists\` and \`all\`, which have no meaning for a single entry. | match | Question answered | |---|---| | (none) | Which keys exist, with a window of values each | | \`{"field":"key","operator":"eq","value":"topics"}\` | Everything about one key | | \`{"field":"key","operator":"prefix","value":"mem-"}\` | A family of keys | -| \`{"field":"key","operator":"in","value":["tags","topics","labels"]}\` | Which of these key names exist | -| \`{"field":"value","operator":"eq","value":"docs-team"}\` | Which keys hold this value (reverse lookup) | -| \`{"field":"value","operator":"prefix","value":"2025-"}\` | Which keys hold values shaped like this | -| \`{"field":"value","operator":"type","value":"boolean"}\` | Which keys hold booleans | +| \`{"field":"value","operator":"eq","value":"docs-team"}\` | Which keys hold this value | | \`{"operator":"and","operands":[{"field":"key","operator":"eq","value":"priority"},{"field":"value","operator":"gte","value":3}]}\` | Values of one key above a threshold | -| \`{"operator":"and","operands":[{"field":"key","operator":"eq","value":"priority"},{"field":"value","operator":"type","value":"number"}]}\` | The numeric side of a key whose documents disagree on type | -Compose with and/or/not to ask several of these in one call. Add \`filter\` to any of them to see what remains after narrowing, e.g. the topics among published documents. The result then reports \`filteredDocuments\`, how many documents pass, and every coverage count is measured against that population. +Add \`filter\` to any of these to see what remains after narrowing. The result then reports \`filteredDocuments\`, and every coverage count is measured against that population. ## Reading the result -\`totalKeys\` keys have a matching entry. \`keys\` holds one window of them ordered by coverage (\`keyLimit\`, default 50, from \`keyOffset\`), and \`remainingKeys\` says how many follow the window. Each key splits by type. Metadata is validated per document, never across documents, so a key can hold numbers in some files and strings in others within a single collection as easily as across collections. A key with more than one type reports each type separately with its own document count and contributing collections, so you can see how many documents a typed filter would reach. Per type: \`documents\` holding it, \`distinctValues\`, one window of \`values\` with document counts (\`valueLimit\`, default 10, from \`valueOffset\`), and \`remainingValues\` after the window. Both remainders are exact. Numbers also report \`range\` (min, median, max) for writing gt/lt thresholds. Counts are documents, not values: a document with \`topics: [a, b]\` counts once for each. - -## Paging - -Page keys with \`keyOffset\` (next page starts at \`keyOffset + keys.length\`) and values with \`valueOffset\`, which applies to every key in the result and so reads best after \`match\` narrows to one key. Raise a limit instead when the remainder is small. +Counts are documents, not values. \`keys\` is one window of \`totalKeys\` in coverage order, and \`remainingKeys\` is the exact count after it. Each key splits by type: metadata is validated per document, so a key can hold numbers in some files and strings in others, and each type reports its own \`documents\`, \`distinctValues\`, a window of \`values\`, \`remainingValues\`, and, for numbers, \`range\` (min, median, max). Page keys with \`keyOffset\` and values with \`valueOffset\`, which applies to every key in the result and so reads best after \`match\` narrows to one key. Every value reported here can be matched with \`{field: '', operator: 'eq', value}\` under the same collections.`, annotations: { readOnlyHint: true, openWorldHint: false }, @@ -759,7 +749,7 @@ Every value reported here can be matched with \`{field: '', operat keyWindowHint: "a higher 'keyLimit' or a 'keyOffset'", keyOffset, keyOffsetLabel: "keyOffset", - emptyMessage: "No metadata matches. Call without match/filter to see every key, or check the status tool for collections with metadata.", + emptyMessage: "No metadata matches. Call without match/filter to see which keys exist, or check the status tool for collections with metadata.", }); return { diff --git a/src/metadata-filter.ts b/src/metadata-filter.ts index c5da72896..fba627583 100644 --- a/src/metadata-filter.ts +++ b/src/metadata-filter.ts @@ -670,7 +670,7 @@ function compileTextTestSql( return `instr(${columnSql}, ?) > 0`; } - params.push(utf8ByteLengthOf(operand), operand); + params.push(Buffer.byteLength(operand, "utf-8"), operand); // SQLite returns NULL for a substring of an empty BLOB. Each primitive // must return a boolean so entry matches and their negations partition rows. return operator === "prefix" @@ -678,12 +678,6 @@ function compileTextTestSql( : `COALESCE(substr(CAST(${columnSql} AS BLOB), -?) = CAST(? AS BLOB), 0)`; } -const utf8Encoder = new TextEncoder(); - -function utf8ByteLengthOf(text: string): number { - return utf8Encoder.encode(text).byteLength; -} - function valueTypeOf(scalar: MetadataScalar): MetadataValueType { return typeof scalar as MetadataValueType; } diff --git a/src/metadata-format.ts b/src/metadata-format.ts index 148140502..351ae0fca 100644 --- a/src/metadata-format.ts +++ b/src/metadata-format.ts @@ -82,7 +82,7 @@ export function formatMetadataKeySummaries(result: ListMetadataResult, options: * body per type. Coverage is measured against the documents the filter * admitted when there is one, otherwise against every active document. */ -export function formatMetadataKeySummary(summary: MetadataKeySummary, result: ListMetadataResult, options: FormatMetadataOptions): string { +function formatMetadataKeySummary(summary: MetadataKeySummary, result: ListMetadataResult, options: FormatMetadataOptions): string { const colors = options.colors ?? NO_COLORS; const typeLabel = summary.types.map(typeLabelOf).join(" | "); const coverage = `${formatCount(summary.documents)} of ${formatCount(result.filteredDocuments ?? result.documents)} documents`; diff --git a/src/metadata-store.ts b/src/metadata-store.ts index 13777ed4a..d97e273b5 100644 --- a/src/metadata-store.ts +++ b/src/metadata-store.ts @@ -507,22 +507,13 @@ function readMetadataReport( } /** - * Key names, coverage, and types for the documents in scope, in coverage - * order, windowed to the first `keyLimit` keys. One GROUP BY, no values: what - * `collection list`, `status`, and the MCP status tool print so a first look - * reveals that metadata exists. `countMetadataKeys` gives the total. + * Every collection's key names, coverage, and types, in coverage order, + * windowed to the first `keyLimit` keys per collection. One GROUP BY, no + * values: what `collection list`, `status`, and the MCP status tool print so + * a first look reveals that metadata exists. */ -export function listMetadataKeys(db: Database, collectionNames?: string[], keyLimit: number = DEFAULT_METADATA_KEY_LIMIT): MetadataKeyOverview[] { - return getMetadataOverview(db, collectionNames, keyLimit).keys; -} - -export function getMetadataOverview(db: Database, collectionNames?: string[], keyLimit: number = DEFAULT_METADATA_KEY_LIMIT): MetadataOverview { - return queryMetadataOverviews(db, buildEligibleCte(collectionNames, undefined), "''", keyLimit).get("") ?? { totalKeys: 0, keys: [] }; -} - -/** Read every collection's overview in one pass, not two queries per collection. */ export function listMetadataCollectionSummaries(db: Database, keyLimit: number = DEFAULT_METADATA_KEY_LIMIT): Map { - return queryMetadataOverviews(db, buildEligibleCte(undefined, undefined), "e.collection", keyLimit); + return queryMetadataOverviews(db, buildEligibleCte(undefined, undefined), keyLimit); } interface MetadataKeyCoverage { @@ -535,22 +526,22 @@ interface MetadataKeyCoverage { total_keys: number; } -function queryMetadataOverviews(db: Database, eligible: Region, collectionSql: string, keyLimit: number): Map { +function queryMetadataOverviews(db: Database, eligible: Region, keyLimit: number): Map { const region = buildRegion(eligible); // Ordinal zero represents a document/key once. Extraction guarantees one // homogeneous type per key, so arrays need neither value reads nor DISTINCT. const coverages = db.prepare(` ${region.withSql}, key_coverages AS ( - SELECT ${collectionSql} AS collection, mv.key, COUNT(*) AS documents, + SELECT e.collection AS collection, mv.key, COUNT(*) AS documents, SUM(mv.value_type = 'string') AS string_documents, SUM(mv.value_type = 'number') AS number_documents, SUM(mv.value_type = 'boolean') AS boolean_documents, - COUNT(*) OVER (PARTITION BY ${collectionSql}) AS total_keys, - ROW_NUMBER() OVER (PARTITION BY ${collectionSql} ORDER BY COUNT(*) DESC, mv.key) AS rank + COUNT(*) OVER (PARTITION BY e.collection) AS total_keys, + ROW_NUMBER() OVER (PARTITION BY e.collection ORDER BY COUNT(*) DESC, mv.key) AS rank ${region.fromSql} WHERE mv.ordinal = 0 - GROUP BY ${collectionSql}, mv.key + GROUP BY e.collection, mv.key ) SELECT * FROM key_coverages WHERE rank <= ? OR ? = -1 ORDER BY collection, rank `).all(...region.withParams, ...region.fromParams, Number.isFinite(keyLimit) ? keyLimit : -1, Number.isFinite(keyLimit) ? keyLimit : -1) as MetadataKeyCoverage[]; diff --git a/test/metadata-cli.test.ts b/test/metadata-cli.test.ts index 9eedf50c9..e99475f35 100644 --- a/test/metadata-cli.test.ts +++ b/test/metadata-cli.test.ts @@ -393,7 +393,7 @@ describe("qmd collection metadata", () => { expect(stdout).toBe([ "filter: 1 of 6 documents", "", - "No metadata matches. Run 'qmd collection metadata' without --match or --filter to see every key.", + "No metadata matches. Run 'qmd collection metadata' without --match or --filter to see which keys exist.", "", ].join("\n")); }, 30000); diff --git a/test/metadata-discovery.test.ts b/test/metadata-discovery.test.ts index 01f13d786..14234222e 100644 --- a/test/metadata-discovery.test.ts +++ b/test/metadata-discovery.test.ts @@ -23,8 +23,6 @@ import { countDocumentsWithMetadata, countMetadataKeys, listMetadata, - listMetadataKeys, - getMetadataOverview, listMetadataCollectionSummaries, METADATA_SQL_BINDING_BUDGET, MetadataBindingBudgetError, @@ -101,7 +99,6 @@ function observeStatements(db: Database, onPrepare: (sql: string) => void, onAll }); } -/** The statement's query plan. Placeholders bind NULL, which is enough to plan. */ function planOf(sql: string): string[] { const placeholders = Array.from(sql.matchAll(/\?/g), () => null); const rows = store.db.prepare(`EXPLAIN QUERY PLAN ${sql}`).all(...placeholders) as { detail: string }[]; @@ -516,8 +513,8 @@ describe("listMetadata key window", () => { expect(fullResult).toEqual(names); expect(pages).toEqual(names); - expect(listMetadataKeys(store.db, ["work"]).map(overview => overview.key)).toEqual(names); - expect(listMetadataKeys(store.db, ["work"], 3).map(overview => overview.key)).toEqual(names.slice(0, 3)); + expect(listMetadataCollectionSummaries(store.db).get("work")!.keys.map(overview => overview.key)).toEqual(names); + expect(listMetadataCollectionSummaries(store.db, 3).get("work")!.keys.map(overview => overview.key)).toEqual(names.slice(0, 3)); }); test("ranks keys once per report and groups values once for both totals and windows", () => { @@ -529,29 +526,29 @@ describe("listMetadata key window", () => { { filter: { field: "e", operator: "prefix", value: "x" }, keyLimit: 5 }, {}, ]; + // One statement ranks the keys, but the planner may still evaluate its + // CTE once per joined row unless it is materialized. Check both, with + // and without table statistics, which change the planner's choice. for (const statistics of [false, true]) { if (statistics) store.db.exec("ANALYZE"); for (const option of options) { statements.length = 0; listMetadata(observed, option); - expect(statements.filter(sql => sql.includes("selected_keys AS"))).toHaveLength(1); + const rankingStatements = statements.filter(sql => sql.includes("selected_keys AS")); + expect(rankingStatements).toHaveLength(1); expect(statements.filter(sql => sql.includes("GROUP BY mv.key, mv.value_type, mv.text_value"))).toHaveLength(1); if (option.filter) expect(statements.some(sql => sql.includes("d.id IN (SELECT mv.document_id"))).toBe(true); - for (const sql of statements) { - const plan = planOf(sql); - expect(plan.some(step => step.includes(" EXISTS ")), sql).toBe(false); - if (sql.includes("d.id IN (SELECT mv.document_id")) expect(plan.some(step => step.includes("LIST SUBQUERY")), sql).toBe(true); - if (!sql.includes("selected_keys AS")) continue; - expect(plan.some(step => step.includes("MATERIALIZE selected_keys")), sql).toBe(true); - expect(plan.some(step => step.includes("CO-ROUTINE selected_keys")), sql).toBe(false); - } + const plan = planOf(rankingStatements[0]!); + expect(plan.some(step => step.includes("MATERIALIZE selected_keys")), rankingStatements[0]).toBe(true); + expect(plan.some(step => step.includes("CO-ROUTINE selected_keys")), rankingStatements[0]).toBe(false); } - statements.length = 0; - listMetadataKeys(observed, undefined, 10); - expect(statements).toHaveLength(1); - expect(statements[0]).toContain("mv.ordinal = 0"); - expect(statements[0]).not.toContain("number_value"); } + + statements.length = 0; + listMetadataCollectionSummaries(observed, 10); + expect(statements).toHaveLength(1); + expect(statements[0]).toContain("mv.ordinal = 0"); + expect(statements[0]).not.toContain("number_value"); }); }); @@ -578,7 +575,6 @@ describe("metadata overviews", () => { const report = listMetadata(store.db, { collection, keyLimit }); const expected = { totalKeys: report.totalKeys, keys: report.keys.map(key => ({ key: key.key, documents: key.documents, types: key.types.map(type => type.type) })) }; expect(overviews.get(collection) ?? { totalKeys: 0, keys: [] }).toEqual(expected); - expect(getMetadataOverview(store.db, [collection], keyLimit)).toEqual(expected); } } } @@ -942,22 +938,21 @@ describe("status view helpers", () => { store.db.prepare(`DELETE FROM document_metadata WHERE document_id = ?`).run(pendingId); }); - test("listMetadataKeys reports names, coverage, and types in coverage order", () => { - expect(listMetadataKeys(store.db)).toEqual([ - { key: "priority", documents: 2, types: ["number", "string"] }, + test("listMetadataCollectionSummaries reports names, coverage, and types per collection in coverage order", () => { + const overviews = listMetadataCollectionSummaries(store.db); + expect(overviews.get("notes")).toEqual({ totalKeys: 2, keys: [ { key: "status", documents: 2, types: ["string"] }, - { key: "source", documents: 1, types: ["string"] }, - ]); - expect(listMetadataKeys(store.db, ["work"])).toEqual([ + { key: "priority", documents: 1, types: ["number"] }, + ] }); + expect(overviews.get("work")).toEqual({ totalKeys: 2, keys: [ { key: "priority", documents: 1, types: ["string"] }, { key: "source", documents: 1, types: ["string"] }, - ]); - expect(listMetadataKeys(store.db, ["missing"])).toEqual([]); - }); + ] }); + expect(overviews.has("missing")).toBe(false); - test("listMetadataKeys windows to the first keys in coverage order", () => { - expect(listMetadataKeys(store.db, undefined, 2).map(overview => overview.key)).toEqual(["priority", "status"]); - expect(listMetadataKeys(store.db, undefined, Infinity)).toHaveLength(3); + const windowed = listMetadataCollectionSummaries(store.db, 1); + expect(windowed.get("notes")).toEqual({ totalKeys: 2, keys: [{ key: "status", documents: 2, types: ["string"] }] }); + expect(listMetadataCollectionSummaries(store.db, Infinity).get("notes")!.keys).toHaveLength(2); }); test("countMetadataKeys reports the vocabulary size in scope", () => { diff --git a/test/metadata-surfaces.test.ts b/test/metadata-surfaces.test.ts index 99f89163e..77d16a027 100644 --- a/test/metadata-surfaces.test.ts +++ b/test/metadata-surfaces.test.ts @@ -463,7 +463,7 @@ describe("MCP and HTTP metadata filter", () => { expect(empty.json.result.isError).toBeFalsy(); expect(empty.json.result.structuredContent).toMatchObject({ documents: 3, filteredDocuments: 0, totalKeys: 0 }); expect(empty.json.result.content[0].text).toBe( - "filter: 0 of 3 documents\n\nNo metadata matches. Call without match/filter to see every key, or check the status tool for collections with metadata.", + "filter: 0 of 3 documents\n\nNo metadata matches. Call without match/filter to see which keys exist, or check the status tool for collections with metadata.", ); const overBudget = await callTool("metadata", buildWidePredicates()); diff --git a/vitest.config.ts b/vitest.config.ts index 415fd4db3..3c95c2419 100644 --- a/vitest.config.ts +++ b/vitest.config.ts @@ -8,6 +8,7 @@ export default defineConfig({ typecheck: { enabled: true, include: ["test/**/*.test-d.ts"], + // Judge only the .test-d.ts assertions; tsc covers src/ separately. ignoreSourceErrors: true, }, }, From 58e89863148ea8b69a80452a030069af7469d6f2 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:38:48 -0500 Subject: [PATCH 24/82] test(bin): pin argv and exit status through the in-process CLI import #923 imports dist/cli/qmd.js in the launcher's own process when the lockfile already selects the running runtime. Its tests check the pid. This adds the parts a user sees: - a Node launch of a Node-selected package runs the CLI in the same pid, with argv aligned to the dist entry plus the user's arguments, and the CLI's exit status reaches the caller; - a Node launch of a Bun-locked package still hands off to a bun child with the dist entry and the same arguments. --- test/bin-wrapper.test.ts | 43 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/test/bin-wrapper.test.ts b/test/bin-wrapper.test.ts index f0592a82f..687118ab9 100644 --- a/test/bin-wrapper.test.ts +++ b/test/bin-wrapper.test.ts @@ -367,6 +367,49 @@ describe("bin/qmd package wrapper", () => { expect(Number(readFileSync(capturePath, "utf8"))).toBe(result.pid); }); + test("an in-process dist import keeps the CLI's argv and exit status", () => { + const { root, capturePath } = makeTempFixture(); + const packageRoot = makePackage(root, "node_modules/@tobilu/qmd"); + const distEntry = join(packageRoot, "dist", "cli", "qmd.js"); + writeFileSync( + distEntry, + guardedEsmCli('writeFileSync(process.env.QMD_WRAPPER_CAPTURE, JSON.stringify({ pid: process.pid, argv: process.argv.slice(1) })); process.exit(3);'), + ); + + const result = spawnSync(REAL_NODE, [join(packageRoot, "bin", "qmd"), "search", "x"], { + env: { ...process.env, QMD_WRAPPER_CAPTURE: capturePath }, + encoding: "utf8", + }); + + expect(result.error).toBeUndefined(); + expect(result.status).toBe(3); + const captured = JSON.parse(readFileSync(capturePath, "utf8")); + expect(captured.pid).toBe(result.pid); + expect(realpathSync(captured.argv[0])).toBe(realpathSync(distEntry)); + expect(captured.argv.slice(1)).toEqual(["search", "x"]); + }); + + test("a Node launch of a Bun-locked package still hands off to a bun child", () => { + const { root, capturePath, runtimeBin } = makeTempFixture(); + const packageRoot = makePackage(root, "node_modules/@tobilu/qmd", ["bun.lock"]); + writeFileSync( + join(packageRoot, "dist", "cli", "qmd.js"), + guardedEsmCli('writeFileSync(process.env.QMD_WRAPPER_CAPTURE, "in-process\\n");'), + ); + + const result = spawnSync(REAL_NODE, [join(packageRoot, "bin", "qmd"), "search", "x"], { + env: { ...process.env, PATH: `${runtimeBin}${delimiter}${process.env.PATH ?? ""}`, QMD_WRAPPER_CAPTURE: capturePath }, + encoding: "utf8", + }); + + expect(result.error).toBeUndefined(); + expect(result.status).toBe(0); + const [runtime, scriptPath, ...args] = readFileSync(capturePath, "utf8").trimEnd().split("\n"); + expect(runtime).toBe("bun"); + expect(realpathSync(scriptPath)).toBe(realpathSync(join(packageRoot, "dist", "cli", "qmd.js"))); + expect(args).toEqual(["search", "x"]); + }); + test("explains how to build when dist is missing and source cannot run", () => { const { root, runtimeBin } = makeTempFixture(); const packageRoot = makePackage(root, "qmd", [], { dist: false }); From 4efbe96464e48f74c7b883d7c5b8828ec47b655d Mon Sep 17 00:00:00 2001 From: naveenspark Date: Wed, 23 Sep 2026 08:22:38 -0500 Subject: [PATCH 25/82] fix: bound doctor vector sampling memory (cherry picked from commit e4bb4110f3b78e0085a015f9d0e0e3e80f1668f3) --- .gitignore | 1 + CHANGELOG.md | 5 ++ DESIGN.md | 10 +++ src/cli/qmd.ts | 12 +--- src/store.ts | 30 +++++++++ test/_helpers/doctor-vector-sample-worker.ts | 30 +++++++++ test/doctor-vector-sampling.test.ts | 65 ++++++++++++++++++++ 7 files changed, 143 insertions(+), 10 deletions(-) create mode 100644 DESIGN.md create mode 100644 test/_helpers/doctor-vector-sample-worker.ts create mode 100644 test/doctor-vector-sampling.test.ts diff --git a/.gitignore b/.gitignore index 165336e8c..ff2c477e8 100644 --- a/.gitignore +++ b/.gitignore @@ -11,6 +11,7 @@ texts/ *.md !README.md !CLAUDE.md +!DESIGN.md !CHANGELOG.md !skills/**/*.md !finetune/*.md diff --git a/CHANGELOG.md b/CHANGELOG.md index ae5792140..5855b0bdf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Fixed + +- `qmd doctor` selects vector sample identities before loading document bodies, + avoiding excessive SQLite memory use on large indexes with duplicate paths. + ### Added - Added Oxlint lint fence. diff --git a/DESIGN.md b/DESIGN.md new file mode 100644 index 000000000..73d75de69 --- /dev/null +++ b/DESIGN.md @@ -0,0 +1,10 @@ +# QMD interface + +QMD is a command-line search tool, library, and MCP server. This change has no +graphical surface. Existing command output and structured result formats are +the observed interface contracts; see README.md and test/cli.test.ts. + +The doctor vector check samples stored chunks and compares freshly generated +embeddings with their saved vectors. Its sampling must preserve active-document, +model, and fingerprint filters without expanding full document bodies across +all candidate chunks. No visual or typography changes are part of this work. diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 318638d87..f22dfa47f 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -28,6 +28,7 @@ import { resolveCommaListName, matchFilesByGlob, getHashesNeedingEmbedding, + getEmbeddingVectorSamples, clearAllEmbeddings, insertEmbedding, getStatus, @@ -4209,16 +4210,7 @@ async function checkEmbeddingVectorSamples(db: Database, model: string, fingerpr return { ok: false, details: "no vector table to test; please run qmd embed again" }; } - const samples = db.prepare(` - SELECT cv.hash, cv.seq, c.doc AS body, MIN(d.path) AS path - FROM content_vectors cv - JOIN documents d ON d.hash = cv.hash AND d.active = 1 - JOIN content c ON c.hash = cv.hash - WHERE cv.model = ? AND cv.embed_fingerprint = ? - GROUP BY cv.hash, cv.seq, c.doc - ORDER BY random() - LIMIT ? - `).all(model, fingerprint, sampleSize) as { hash: string; seq: number; body: string; path: string }[]; + const samples = getEmbeddingVectorSamples(db, model, fingerprint, sampleSize); if (samples.length === 0) { return { ok: false, details: "no current embedded chunks to test; please run qmd embed again" }; diff --git a/src/store.ts b/src/store.ts index 2c719da1e..932ef5e72 100644 --- a/src/store.ts +++ b/src/store.ts @@ -2563,6 +2563,36 @@ export type IndexStatusSummary = Omit & { // Index health // ============================================================================= +export type EmbeddingVectorSample = { + hash: string; + seq: number; + body: string; + path: string; +}; + +export function getEmbeddingVectorSamples(db: Database, model: string, fingerprint: string, sampleSize: number = 3): EmbeddingVectorSample[] { + // Limit chunk identities before reading bodies. Joining bodies to every chunk + // and duplicate path can make a three-row sample sort gigabytes of text. + return db.prepare(` + WITH sampled AS MATERIALIZED ( + SELECT cv.hash, cv.seq + FROM content_vectors cv + JOIN content c ON c.hash = cv.hash + WHERE cv.model = ? AND cv.embed_fingerprint = ? + AND EXISTS ( + SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1 + ) + ORDER BY random() + LIMIT ? + ) + SELECT sampled.hash, sampled.seq, c.doc AS body, + (SELECT MIN(d.path) FROM documents d + WHERE d.hash = sampled.hash AND d.active = 1) AS path + FROM sampled + JOIN content c ON c.hash = sampled.hash + `).all(model, fingerprint, sampleSize); +} + export function getHashesNeedingEmbedding(db: Database, collection?: string, model: string = DEFAULT_EMBED_MODEL): number { const collectionFilter = collection ? `AND d.collection = ?` : ``; const fingerprint = getEmbeddingFingerprint(model); diff --git a/test/_helpers/doctor-vector-sample-worker.ts b/test/_helpers/doctor-vector-sample-worker.ts new file mode 100644 index 000000000..3c2135949 --- /dev/null +++ b/test/_helpers/doctor-vector-sample-worker.ts @@ -0,0 +1,30 @@ +import { createStore, getEmbeddingVectorSamples } from "../../src/store.js"; + +const store = createStore(":memory:"); +try { + const body = "word ".repeat(13_107); // About 64 KiB per document. + const insertChunk = store.db.prepare(` + INSERT INTO content_vectors (hash, seq, model, embed_fingerprint, embedded_at) + VALUES (?, ?, 'model', 'current', '2026-01-01') + `); + store.db.transaction(() => { + for (let doc = 0; doc < 16; doc++) { + const hash = `document-${doc}`; + store.insertContent(hash, body, "2026-01-01"); + for (let path = 0; path < 8; path++) { + store.insertDocument("test", `${hash}-${path}.md`, hash, hash, "2026-01-01", "2026-01-01"); + } + for (let seq = 0; seq < 32; seq++) insertChunk.run(hash, seq); + } + })(); + store.db.exec("PRAGMA temp_store = MEMORY"); + store.db.exec("PRAGMA hard_heap_limit = 33554432"); + const samples = getEmbeddingVectorSamples(store.db, "model", "current"); + console.log(JSON.stringify({ + samples: samples.length, + distinctChunks: new Set(samples.map(sample => `${sample.hash}:${sample.seq}`)).size, + bodiesComplete: samples.every(sample => sample.body === body), + })); +} finally { + store.close(); +} diff --git a/test/doctor-vector-sampling.test.ts b/test/doctor-vector-sampling.test.ts new file mode 100644 index 000000000..ac19ff3fb --- /dev/null +++ b/test/doctor-vector-sampling.test.ts @@ -0,0 +1,65 @@ +import { afterEach, beforeEach, describe, expect, test } from "vitest"; +import { spawnSync } from "node:child_process"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { isBun } from "../src/db.js"; +import { createStore, getEmbeddingVectorSamples, type Store } from "../src/store.js"; + +const projectRoot = join(dirname(fileURLToPath(import.meta.url)), ".."); +let store: Store; + +beforeEach(() => { store = createStore(":memory:"); }); +afterEach(() => { store.close(); }); + +function addDocument(hash: string, path: string, active: boolean = true): void { + store.insertContent(hash, `Body for ${hash}`, "2026-01-01"); + const id = store.insertDocument("test", path, hash, hash, "2026-01-01", "2026-01-01"); + if (!active) store.db.prepare("UPDATE documents SET active = 0 WHERE id = ?").run(id); +} + +function addChunk(hash: string, seq: number, model: string = "model", fingerprint: string = "current"): void { + store.db.prepare(` + INSERT INTO content_vectors (hash, seq, model, embed_fingerprint, embedded_at) + VALUES (?, ?, ?, ?, '2026-01-01') + `).run(hash, seq, model, fingerprint); +} + +describe("doctor vector sampling", () => { + test("samples each eligible chunk once and uses the first active path", () => { + addDocument("shared", "z.md"); + addDocument("shared", "b.md"); + addDocument("shared", "a.md", false); + addChunk("shared", 0); + addChunk("shared", 1); + addDocument("other", "other.md"); + addChunk("other", 0); + addDocument("inactive", "inactive.md", false); + addChunk("inactive", 0); + addDocument("stale", "stale.md"); + addChunk("stale", 0, "model", "old"); + addDocument("different-model", "different.md"); + addChunk("different-model", 0, "other-model"); + addChunk("orphan", 0); + + const samples = getEmbeddingVectorSamples(store.db, "model", "current", 20); + expect(samples.sort((a, b) => a.hash.localeCompare(b.hash) || a.seq - b.seq)).toEqual([ + { hash: "other", seq: 0, body: "Body for other", path: "other.md" }, + { hash: "shared", seq: 0, body: "Body for shared", path: "b.md" }, + { hash: "shared", seq: 1, body: "Body for shared", path: "b.md" }, + ]); + expect(getEmbeddingVectorSamples(store.db, "model", "current", 2)).toHaveLength(2); + expect(getEmbeddingVectorSamples(store.db, "model", "current", 0)).toEqual([]); + expect(getEmbeddingVectorSamples(store.db, "missing", "current")).toEqual([]); + }); + + test("samples large documents with duplicate paths within a 32 MiB SQLite budget", () => { + // SQLite's hard heap limit is process-wide and cannot be raised again. + const worker = join(projectRoot, "test", "_helpers", "doctor-vector-sample-worker.ts"); + const args = isBun ? [worker] : [join(projectRoot, "node_modules", "tsx", "dist", "cli.mjs"), worker]; + const result = spawnSync(process.execPath, args, { encoding: "utf8", timeout: 20_000 }); + expect(result.error).toBeUndefined(); + expect(result.stderr).toBe(""); + expect(result.status).toBe(0); + expect(JSON.parse(result.stdout)).toEqual({ samples: 3, distinctChunks: 3, bodiesComplete: true }); + }); +}); From 8d43a232f7b65ab9af9687414944e0d97c6014ff Mon Sep 17 00:00:00 2001 From: naveenspark Date: Wed, 23 Sep 2026 08:24:13 -0500 Subject: [PATCH 26/82] test: deactivate sampling fixtures through the store API (cherry picked from commit ace57bd2c0440519f8dfc9c25b7a7d6e240b815b) --- test/doctor-vector-sampling.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/test/doctor-vector-sampling.test.ts b/test/doctor-vector-sampling.test.ts index ac19ff3fb..736c7655f 100644 --- a/test/doctor-vector-sampling.test.ts +++ b/test/doctor-vector-sampling.test.ts @@ -13,8 +13,8 @@ afterEach(() => { store.close(); }); function addDocument(hash: string, path: string, active: boolean = true): void { store.insertContent(hash, `Body for ${hash}`, "2026-01-01"); - const id = store.insertDocument("test", path, hash, hash, "2026-01-01", "2026-01-01"); - if (!active) store.db.prepare("UPDATE documents SET active = 0 WHERE id = ?").run(id); + store.insertDocument("test", path, hash, hash, "2026-01-01", "2026-01-01"); + if (!active) store.deactivateDocument("test", path); } function addChunk(hash: string, seq: number, model: string = "model", fingerprint: string = "current"): void { From 2e8b8e68d94f8f22e39cc4e96355a224b9f6027c Mon Sep 17 00:00:00 2001 From: naveenspark Date: Wed, 23 Sep 2026 11:48:11 -0500 Subject: [PATCH 27/82] fix: compare doctor vectors at their stored positions (cherry picked from commit 0bd6c19ee1a0bf081eb610d51f63df35696f74f2) --- CHANGELOG.md | 2 + DESIGN.md | 2 + src/cli/qmd.ts | 6 ++- src/store.ts | 5 ++- test/doctor-vector-sampling.test.ts | 67 ++++++++++++++++++++++++----- 5 files changed, 68 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5855b0bdf..9aae599e4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,8 @@ - `qmd doctor` selects vector sample identities before loading document bodies, avoiding excessive SQLite memory use on large indexes with duplicate paths. +- Vector diagnostics match passages by their saved character position, avoiding + false mismatches when earlier chunks change the sequence numbering. ### Added diff --git a/DESIGN.md b/DESIGN.md index 73d75de69..618822490 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -8,3 +8,5 @@ The doctor vector check samples stored chunks and compares freshly generated embeddings with their saved vectors. Its sampling must preserve active-document, model, and fingerprint filters without expanding full document bodies across all candidate chunks. No visual or typography changes are part of this work. +The saved character position identifies the passage to re-embed; a stored +sequence number is the vector key and may differ from today's chunk ordering. diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index f22dfa47f..2c074c609 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -4199,7 +4199,7 @@ function checkModelCache(activeModels: { embed: string; generate: string; rerank } } -async function checkEmbeddingVectorSamples(db: Database, model: string, fingerprint: string, sampleSize: number = 3): Promise { +export async function checkEmbeddingVectorSamples(db: Database, model: string, fingerprint: string, sampleSize: number = 3): Promise { const activeDocs = (db.prepare(`SELECT COUNT(*) AS count FROM documents WHERE active = 1`).get() as { count: number }).count; if (activeDocs === 0) { return { ok: true, details: "no active documents indexed" }; @@ -4223,7 +4223,9 @@ async function checkEmbeddingVectorSamples(db: Database, model: string, fingerpr for (const sample of samples) { const hashSeq = `${sample.hash}_${sample.seq}`; const chunks = await chunkDocumentByTokens(sample.body, undefined, undefined, undefined, sample.path, undefined, session.signal); - const chunk = chunks[sample.seq]; + // Sequence numbers identify stored vectors, but earlier chunks can split + // differently after a tokenizer/chunker change. Compare the saved passage. + const chunk = chunks.find(chunk => chunk.pos === sample.pos); if (!chunk) { mismatches.push(`${shortHashSeq(hashSeq)}: chunk no longer exists`); continue; diff --git a/src/store.ts b/src/store.ts index 932ef5e72..ff5f76677 100644 --- a/src/store.ts +++ b/src/store.ts @@ -2566,6 +2566,7 @@ export type IndexStatusSummary = Omit & { export type EmbeddingVectorSample = { hash: string; seq: number; + pos: number; body: string; path: string; }; @@ -2575,7 +2576,7 @@ export function getEmbeddingVectorSamples(db: Database, model: string, fingerpri // and duplicate path can make a three-row sample sort gigabytes of text. return db.prepare(` WITH sampled AS MATERIALIZED ( - SELECT cv.hash, cv.seq + SELECT cv.hash, cv.seq, cv.pos FROM content_vectors cv JOIN content c ON c.hash = cv.hash WHERE cv.model = ? AND cv.embed_fingerprint = ? @@ -2585,7 +2586,7 @@ export function getEmbeddingVectorSamples(db: Database, model: string, fingerpri ORDER BY random() LIMIT ? ) - SELECT sampled.hash, sampled.seq, c.doc AS body, + SELECT sampled.hash, sampled.seq, sampled.pos, c.doc AS body, (SELECT MIN(d.path) FROM documents d WHERE d.hash = sampled.hash AND d.active = 1) AS path FROM sampled diff --git a/test/doctor-vector-sampling.test.ts b/test/doctor-vector-sampling.test.ts index 736c7655f..b563b98fa 100644 --- a/test/doctor-vector-sampling.test.ts +++ b/test/doctor-vector-sampling.test.ts @@ -3,13 +3,15 @@ import { spawnSync } from "node:child_process"; import { dirname, join } from "node:path"; import { fileURLToPath } from "node:url"; import { isBun } from "../src/db.js"; -import { createStore, getEmbeddingVectorSamples, type Store } from "../src/store.js"; +import { chunkDocumentByTokens, createStore, extractTitle, formatDocForEmbedding, getEmbeddingVectorSamples, type Store } from "../src/store.js"; +import { checkEmbeddingVectorSamples } from "../src/cli/qmd.js"; +import { setDefaultLlamaCpp, LlamaCpp } from "../src/llm.js"; const projectRoot = join(dirname(fileURLToPath(import.meta.url)), ".."); let store: Store; beforeEach(() => { store = createStore(":memory:"); }); -afterEach(() => { store.close(); }); +afterEach(() => { setDefaultLlamaCpp(null); store.close(); }); function addDocument(hash: string, path: string, active: boolean = true): void { store.insertContent(hash, `Body for ${hash}`, "2026-01-01"); @@ -17,20 +19,65 @@ function addDocument(hash: string, path: string, active: boolean = true): void { if (!active) store.deactivateDocument("test", path); } -function addChunk(hash: string, seq: number, model: string = "model", fingerprint: string = "current"): void { +function addChunk(hash: string, seq: number, model: string = "model", fingerprint: string = "current", pos: number = 0): void { store.db.prepare(` - INSERT INTO content_vectors (hash, seq, model, embed_fingerprint, embedded_at) - VALUES (?, ?, ?, ?, '2026-01-01') - `).run(hash, seq, model, fingerprint); + INSERT INTO content_vectors (hash, seq, model, embed_fingerprint, pos, embedded_at) + VALUES (?, ?, ?, ?, ?, '2026-01-01') + `).run(hash, seq, model, fingerprint, pos); } describe("doctor vector sampling", () => { + async function addStoredPassage(seq: number, positionExists: boolean = true, vectorMatches: boolean = true): Promise { + let expectedText = ""; + class PassageLlm extends LlamaCpp { + async tokenize(text: string) { return new Array(Math.ceil(text.length / 16)).fill(1); } + async embed(text: string) { + return { embedding: text === expectedText ? [1, 0] : [0, 1], model: "model" }; + } + } + setDefaultLlamaCpp(new PassageLlm()); + const body = "First passage. ".repeat(300) + "\n\nSecond passage. ".repeat(300); + const chunks = await chunkDocumentByTokens(body); + const passage = chunks[1]!; + expect(passage.pos).toBeGreaterThan(0); + expectedText = formatDocForEmbedding(passage.text, extractTitle(body, "sample.md"), "model"); + store.insertContent("sample", body, "2026-01-01"); + store.insertDocument("test", "sample.md", "Sample", "sample", "2026-01-01", "2026-01-01"); + store.ensureVecTable(2); + store.insertEmbedding("sample", seq, positionExists ? passage.pos : 1, + new Float32Array(vectorMatches ? [1, 0] : [0, 1]), "model", "2026-01-01", 1, "current"); + } + + test("checks the stored passage when earlier chunks shift its sequence number", async () => { + await addStoredPassage(0); + expect((await checkEmbeddingVectorSamples(store.db, "model", "current")).ok).toBe(true); + }); + + test("checks the stored passage when its old sequence is beyond today's chunk count", async () => { + await addStoredPassage(100); + expect((await checkEmbeddingVectorSamples(store.db, "model", "current")).ok).toBe(true); + }); + + test("still rejects a changed vector at the matching position", async () => { + await addStoredPassage(1, true, false); + const result = await checkEmbeddingVectorSamples(store.db, "model", "current"); + expect(result.ok).toBe(false); + expect(result.details).toContain("stored vector distance"); + }); + + test("does not substitute the sequence number when the stored position disappeared", async () => { + await addStoredPassage(1, false); + const result = await checkEmbeddingVectorSamples(store.db, "model", "current"); + expect(result.ok).toBe(false); + expect(result.details).toContain("chunk no longer exists"); + }); + test("samples each eligible chunk once and uses the first active path", () => { addDocument("shared", "z.md"); addDocument("shared", "b.md"); addDocument("shared", "a.md", false); addChunk("shared", 0); - addChunk("shared", 1); + addChunk("shared", 1, "model", "current", 1234); addDocument("other", "other.md"); addChunk("other", 0); addDocument("inactive", "inactive.md", false); @@ -43,9 +90,9 @@ describe("doctor vector sampling", () => { const samples = getEmbeddingVectorSamples(store.db, "model", "current", 20); expect(samples.sort((a, b) => a.hash.localeCompare(b.hash) || a.seq - b.seq)).toEqual([ - { hash: "other", seq: 0, body: "Body for other", path: "other.md" }, - { hash: "shared", seq: 0, body: "Body for shared", path: "b.md" }, - { hash: "shared", seq: 1, body: "Body for shared", path: "b.md" }, + { hash: "other", seq: 0, pos: 0, body: "Body for other", path: "other.md" }, + { hash: "shared", seq: 0, pos: 0, body: "Body for shared", path: "b.md" }, + { hash: "shared", seq: 1, pos: 1234, body: "Body for shared", path: "b.md" }, ]); expect(getEmbeddingVectorSamples(store.db, "model", "current", 2)).toHaveLength(2); expect(getEmbeddingVectorSamples(store.db, "model", "current", 0)).toEqual([]); From 2e5592769b98a2b03b973a1071c8215892da3942 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 3 Sep 2026 12:51:13 -0500 Subject: [PATCH 28/82] feat(store): vector layout resolver with integer collection ids Add src/vec-layout.ts as the one place that knows which sqlite-vec table is live and what it is called. It resolves `none`, the legacy `vectors_vec` (hash_seq or hash keyed) or the partitioned `vectors_by_collection`, with the vec0 shadow-table names and the stored dimension, and it owns the two plain tables the partitioned layout needs: `vector_collection_ids` (a stable integer id per collection name, so a rename is one row) and `vector_rows` (vec0 rowid to hash, seq, collection_id, NOT NULL and UNIQUE per triple). Both are created at open next to `content_vectors`. The resolver also carries the collection id allocate, resolve, rename and delete helpers and the JSON rowid list used to bind a scoped search's rowids through `json_each` in one parameter. Nothing reads the partitioned table yet; the legacy table stays live. (cherry picked from commit 3cdb92dc822364f16ae1d5cd01ba6f91dcdd4c4a) --- src/store.ts | 2 + src/vec-layout.ts | 167 ++++++++++++++++++++++++++++++++++++++++ test/vec-layout.test.ts | 157 +++++++++++++++++++++++++++++++++++++ 3 files changed, 326 insertions(+) create mode 100644 src/vec-layout.ts create mode 100644 test/vec-layout.test.ts diff --git a/src/store.ts b/src/store.ts index 6cdada20f..3617115ac 100644 --- a/src/store.ts +++ b/src/store.ts @@ -12,6 +12,7 @@ */ import { openDatabase, loadSqliteVec } from "./db.js"; +import { createVectorMetadataTables } from "./vec-layout.js"; import type { Database } from "./db.js"; import picomatch from "picomatch"; import { createHash } from "crypto"; @@ -1275,6 +1276,7 @@ function initializeDatabase(db: Database): void { `); ensureContentVectorsStatusIndex(db); + createVectorMetadataTables(db); // Document metadata — extraction state plus normalized value rows for // metadata filtering. Keyed by document identity, not content hash. diff --git a/src/vec-layout.ts b/src/vec-layout.ts new file mode 100644 index 000000000..704d9aae5 --- /dev/null +++ b/src/vec-layout.ts @@ -0,0 +1,167 @@ +/** + * vec-layout.ts - Names and resolution of the sqlite-vec vector layout. + * + * Vectors live in one vec0 table partitioned by an integer collection id, one + * row per (chunk, active collection). `vector_rows` maps each vec0 rowid back + * to its (hash, seq, collection_id), and `vector_collection_ids` maps names to + * ids. The integer key is what makes `qmd collection rename` a one-row update: + * vec0 rejects UPDATE on a partition key, so a name-keyed partition would have + * to rewrite every row of the collection. + * + * A layout change needs a new permanent table name: `ALTER TABLE RENAME` on a + * vec0 table leaves its shadow tables under the old name (sqlite-vec 0.1.9), + * and SQLite refuses to rename shadow tables by hand. `vecLayout` is the one + * place that knows which table is live, so no other module spells the names. + */ + +import type { Database } from "./db.js"; + +export const VEC_TABLE = "vectors_by_collection"; +export const VEC_ROWS_TABLE = "vector_rows"; +export const VEC_COLLECTION_IDS_TABLE = "vector_collection_ids"; +export const LEGACY_VEC_TABLE = "vectors_vec"; + +/** vec0's shadow tables for a virtual table name. */ +export type VecShadowTables = { + chunks: string; + rowids: string; + vectorChunks: string; + info: string; +}; + +export type VecLayout = + | { kind: "none" } + | { kind: "legacy"; table: typeof LEGACY_VEC_TABLE; dimensions: number | null; keyedByHashSeq: boolean; shadow: VecShadowTables } + | { kind: "partitioned"; table: typeof VEC_TABLE; dimensions: number | null; shadow: VecShadowTables }; + +export type ReadableVecLayout = Exclude; + +function shadowTables(table: string): VecShadowTables { + return { + chunks: `${table}_chunks`, + rowids: `${table}_rowids`, + vectorChunks: `${table}_vector_chunks00`, + info: `${table}_info`, + }; +} + +function parseDimensions(sql: string | null): number | null { + const match = sql?.match(/float\[(\d+)\]/); + return match?.[1] ? parseInt(match[1], 10) : null; +} + +/** + * Which vector layout the database holds. A legacy table wins while it + * exists: the migration drops it as its last step, so that drop is the flip. + */ +export function vecLayout(db: Database): VecLayout { + const rows = db.prepare( + `SELECT name, sql FROM sqlite_master WHERE type = 'table' AND name IN (?, ?)` + ).all(LEGACY_VEC_TABLE, VEC_TABLE) as { name: string; sql: string | null }[]; + + const legacy = rows.find((row) => row.name === LEGACY_VEC_TABLE); + if (legacy) { + return { + kind: "legacy", + table: LEGACY_VEC_TABLE, + dimensions: parseDimensions(legacy.sql), + keyedByHashSeq: (legacy.sql ?? "").includes("hash_seq"), + shadow: shadowTables(LEGACY_VEC_TABLE), + }; + } + const partitioned = rows.find((row) => row.name === VEC_TABLE); + if (partitioned) { + return { + kind: "partitioned", + table: VEC_TABLE, + dimensions: parseDimensions(partitioned.sql), + shadow: shadowTables(VEC_TABLE), + }; + } + return { kind: "none" }; +} + +/** True when the partitioned vector index exists; vector search reads nothing else. */ +export function hasVectorIndex(db: Database): boolean { + return vecLayout(db).kind === "partitioned"; +} + +/** + * A schema entry can outlive the vec0 module (reopening without sqlite-vec + * loaded), in which case touching the table throws "no such module". + */ +export function vecTableReadable(db: Database, layout: ReadableVecLayout): boolean { + try { + db.prepare(`SELECT 1 FROM ${layout.table} LIMIT 0`).get(); + return true; + } catch { + return false; + } +} + +export function createVectorMetadataTables(db: Database): void { + db.exec(` + CREATE TABLE IF NOT EXISTS ${VEC_COLLECTION_IDS_TABLE} ( + id INTEGER PRIMARY KEY, + name TEXT NOT NULL UNIQUE + ) + `); + db.exec(` + CREATE TABLE IF NOT EXISTS ${VEC_ROWS_TABLE} ( + id INTEGER PRIMARY KEY, + hash TEXT NOT NULL, + seq INTEGER NOT NULL, + collection_id INTEGER NOT NULL, + UNIQUE(hash, seq, collection_id) + ) + `); + db.exec(`CREATE INDEX IF NOT EXISTS idx_vector_rows_collection ON ${VEC_ROWS_TABLE}(collection_id)`); +} + +export function createPartitionedVecTable(db: Database, dimensions: number): void { + db.exec( + `CREATE VIRTUAL TABLE ${VEC_TABLE} USING vec0(collection_id INTEGER PARTITION KEY, embedding float[${dimensions}] distance_metric=cosine)` + ); +} + +export function resolveCollectionId(db: Database, name: string): number | undefined { + const row = db.prepare(`SELECT id FROM ${VEC_COLLECTION_IDS_TABLE} WHERE name = ?`).get(name) as { id: number } | undefined; + return row?.id; +} + +/** Ids of the named collections that have one; names without an id are absent. */ +export function resolveCollectionIds(db: Database, names: readonly string[]): Map { + const ids = new Map(); + const stmt = db.prepare(`SELECT id FROM ${VEC_COLLECTION_IDS_TABLE} WHERE name = ?`); + for (const name of names) { + const row = stmt.get(name) as { id: number } | undefined; + if (row) ids.set(name, row.id); + } + return ids; +} + +export function allocateCollectionId(db: Database, name: string): number { + db.prepare(`INSERT OR IGNORE INTO ${VEC_COLLECTION_IDS_TABLE} (name) VALUES (?)`).run(name); + const id = resolveCollectionId(db, name); + if (id === undefined || id === null) { + throw new Error(`Could not allocate a vector collection id for '${name}'`); + } + return id; +} + +export function renameCollectionId(db: Database, oldName: string, newName: string): void { + db.prepare(`UPDATE ${VEC_COLLECTION_IDS_TABLE} SET name = ? WHERE name = ?`).run(newName, oldName); +} + +export function deleteCollectionId(db: Database, name: string): void { + db.prepare(`DELETE FROM ${VEC_COLLECTION_IDS_TABLE} WHERE name = ?`).run(name); +} + +/** + * One bound parameter for `json_each(?)`: a scoped search over many + * partitions can return more rowids than SQLite allows as separate + * parameters (32,766). + */ +export function rowidList(rowids: readonly number[]): string { + return JSON.stringify(rowids); +} diff --git a/test/vec-layout.test.ts b/test/vec-layout.test.ts new file mode 100644 index 000000000..110d9c132 --- /dev/null +++ b/test/vec-layout.test.ts @@ -0,0 +1,157 @@ +/** + * The vector layout resolver owns the names of the sqlite-vec tables and the + * integer collection ids that partition the vec0 table. + */ +import { describe, test, expect, afterEach } from "vitest"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createStore, type Store } from "../src/store.js"; +import { + LEGACY_VEC_TABLE, + VEC_COLLECTION_IDS_TABLE, + VEC_ROWS_TABLE, + VEC_TABLE, + allocateCollectionId, + createPartitionedVecTable, + hasVectorIndex, + renameCollectionId, + resolveCollectionId, + resolveCollectionIds, + rowidList, + vecLayout, + vecTableReadable, +} from "../src/vec-layout.js"; + +let store: Store | null = null; +let dir: string | null = null; + +async function openStore(): Promise { + dir = await mkdtemp(join(tmpdir(), "qmd-vec-layout-")); + store = createStore(join(dir, "index.sqlite")); + return store; +} + +afterEach(async () => { + store?.close(); + store = null; + if (dir) await rm(dir, { recursive: true, force: true }); + dir = null; +}); + +function tableExists(s: Store, name: string): boolean { + return Boolean(s.db.prepare(`SELECT name FROM sqlite_master WHERE type = 'table' AND name = ?`).get(name)); +} + +describe("vecLayout", () => { + test("resolves none on a store without a vector table", async () => { + const s = await openStore(); + expect(vecLayout(s.db)).toEqual({ kind: "none" }); + expect(hasVectorIndex(s.db)).toBe(false); + }); + + test("creates the collection id and row mapping tables at open", async () => { + const s = await openStore(); + expect(tableExists(s, VEC_COLLECTION_IDS_TABLE)).toBe(true); + expect(tableExists(s, VEC_ROWS_TABLE)).toBe(true); + }); + + test("resolves the partitioned table with its dimensions and shadow tables", async () => { + const s = await openStore(); + createPartitionedVecTable(s.db, 3); + + const layout = vecLayout(s.db); + expect(layout.kind).toBe("partitioned"); + if (layout.kind !== "partitioned") return; + expect(layout.table).toBe(VEC_TABLE); + expect(layout.dimensions).toBe(3); + expect(layout.shadow).toEqual({ + chunks: `${VEC_TABLE}_chunks`, + rowids: `${VEC_TABLE}_rowids`, + vectorChunks: `${VEC_TABLE}_vector_chunks00`, + info: `${VEC_TABLE}_info`, + }); + expect(tableExists(s, layout.shadow.chunks)).toBe(true); + expect(tableExists(s, layout.shadow.rowids)).toBe(true); + expect(hasVectorIndex(s.db)).toBe(true); + expect(vecTableReadable(s.db, layout)).toBe(true); + }); + + test("resolves a legacy hash_seq table, and keeps resolving it while both tables exist", async () => { + const s = await openStore(); + s.db.exec(`CREATE VIRTUAL TABLE ${LEGACY_VEC_TABLE} USING vec0(hash_seq TEXT PRIMARY KEY, embedding float[4] distance_metric=cosine)`); + + const legacy = vecLayout(s.db); + expect(legacy).toMatchObject({ kind: "legacy", table: LEGACY_VEC_TABLE, dimensions: 4, keyedByHashSeq: true }); + expect(hasVectorIndex(s.db)).toBe(false); + + createPartitionedVecTable(s.db, 4); + expect(vecLayout(s.db).kind).toBe("legacy"); + }); + + test("marks a legacy table keyed by hash alone", async () => { + const s = await openStore(); + s.db.exec(`CREATE VIRTUAL TABLE ${LEGACY_VEC_TABLE} USING vec0(hash TEXT PRIMARY KEY, embedding float[2] distance_metric=cosine)`); + expect(vecLayout(s.db)).toMatchObject({ kind: "legacy", dimensions: 2, keyedByHashSeq: false }); + }); + + test("reports a schema entry whose module is not loaded as unreadable", async () => { + const s = await openStore(); + s.db.exec(`CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`); + const layout = vecLayout(s.db); + expect(layout.kind).toBe("partitioned"); + if (layout.kind !== "partitioned") return; + expect(layout.dimensions).toBeNull(); + expect(vecTableReadable(s.db, layout)).toBe(true); + s.db.exec(`DROP TABLE ${VEC_TABLE}`); + expect(vecTableReadable(s.db, layout)).toBe(false); + }); +}); + +describe("collection ids", () => { + test("allocates a stable integer id per name and resolves it back", async () => { + const s = await openStore(); + const docs = allocateCollectionId(s.db, "docs"); + const notes = allocateCollectionId(s.db, "notes"); + expect(docs).not.toBe(notes); + expect(allocateCollectionId(s.db, "docs")).toBe(docs); + expect(resolveCollectionId(s.db, "docs")).toBe(docs); + expect(resolveCollectionId(s.db, "unknown")).toBeUndefined(); + expect(resolveCollectionIds(s.db, ["notes", "unknown", "docs"])).toEqual(new Map([["docs", docs], ["notes", notes]])); + }); + + test("rename keeps the id so partition rows stay reachable", async () => { + const s = await openStore(); + const id = allocateCollectionId(s.db, "old"); + renameCollectionId(s.db, "old", "new"); + expect(resolveCollectionId(s.db, "old")).toBeUndefined(); + expect(resolveCollectionId(s.db, "new")).toBe(id); + }); + + test("rename of a name without an id is a no-op", async () => { + const s = await openStore(); + expect(() => renameCollectionId(s.db, "missing", "other")).not.toThrow(); + expect(resolveCollectionId(s.db, "other")).toBeUndefined(); + }); + + test("the row mapping refuses a null collection id", async () => { + const s = await openStore(); + expect(() => s.db.prepare(`INSERT INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES ('h', 0, NULL)`).run()).toThrow(/NOT NULL/); + }); + + test("the row mapping is unique per (hash, seq, collection)", async () => { + const s = await openStore(); + const id = allocateCollectionId(s.db, "docs"); + s.db.prepare(`INSERT INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES ('h', 0, ?)`).run(id); + expect(() => s.db.prepare(`INSERT INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES ('h', 0, ?)`).run(id)).toThrow(/UNIQUE/); + }); +}); + +describe("rowid lists", () => { + test("json_each expands a rowid list on the loaded SQLite build", async () => { + const s = await openStore(); + const ids = Array.from({ length: 60_000 }, (_, i) => i + 1); + const row = s.db.prepare(`SELECT COUNT(*) AS n, MAX(value) AS top FROM json_each(?)`).get(rowidList(ids)) as { n: number; top: number }; + expect(row).toEqual({ n: 60_000, top: 60_000 }); + }); +}); From e1642b7ea1625ee9de31e2b7408c7651b1d3aa85 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 3 Sep 2026 12:57:39 -0500 Subject: [PATCH 29/82] feat(store): migration step that copies legacy vectors into per-collection partitions Add src/store-migrations.ts with the PRAGMA user_version ladder: version 1 installs the FTS sync triggers as before, and version 2 copies every row of the legacy `vectors_vec` table into the partitioned layout, one vec0 row per (chunk, active collection), keyed by the integer collection id. Nothing calls the step at open yet; that lands with the readers and writers that use the new layout. The copy reads the legacy table's chunk blobs directly (validity bitmap, rowid list, packed float32 vectors) at one chunk per IMMEDIATE transaction and keeps the last copied chunk_id in store_config. A process killed between chunks resumes from that cursor, and two processes opening the same database take turns, because each reads the cursor inside its own write transaction and stops as soon as the other stamps the version. A verification pass then copies by key any chunk that content_vectors and an active document claim but no partition holds (a pre-upgrade repack can move a row into a chunk the walk already passed), content_vectors rows with no vector anywhere are deleted so the pending detector embeds them again, and the version stamp and the DROP of the legacy table share one transaction. A legacy table keyed by hash alone is dropped without a copy, as before. Without sqlite-vec the step defers and the version stays at 1. vec0 checks the bound type of its partition key and rowid rather than applying affinity, and better-sqlite3 binds every JS number as a double, so those parameters are bound as bigint through `vecInteger`. The anti-join that finds chunks missing from a partition lives in vec-layout.ts because the embed path will need the same query. (cherry picked from commit 7d1735494e61f16c0991eb247f0d432a93141324) --- src/store-migrations.ts | 297 ++++++++++++++++++++++++++++++ src/vec-layout.ts | 30 ++++ test/store-migrations.test.ts | 328 ++++++++++++++++++++++++++++++++++ 3 files changed, 655 insertions(+) create mode 100644 src/store-migrations.ts create mode 100644 test/store-migrations.test.ts diff --git a/src/store-migrations.ts b/src/store-migrations.ts new file mode 100644 index 000000000..5bfbedb64 --- /dev/null +++ b/src/store-migrations.ts @@ -0,0 +1,297 @@ +/** + * store-migrations.ts - PRAGMA user_version steps applied when a store opens. + * + * Version 1 installs the FTS sync triggers. Version 2 moves every vector out + * of the legacy `vectors_vec` table (one row per chunk, keyed by hash_seq) + * into the layout `vec-layout.ts` describes: one row per (chunk, active + * collection) in a vec0 table partitioned by collection id. + * + * The copy walks the legacy table's chunk blobs in chunk_id order, one chunk + * per IMMEDIATE transaction, and keeps the last copied chunk_id in + * store_config. A killed process resumes from that cursor, and two processes + * that open the same database take turns on chunks because each reads the + * cursor inside its own write transaction. The legacy table is dropped in the + * same transaction that stamps the version, so that drop is the flip. + */ + +import type { Database } from "./db.js"; +import { + LEGACY_VEC_TABLE, + VEC_ROWS_TABLE, + VEC_TABLE, + allocateCollectionId, + createPartitionedVecTable, + missingPartitionRows, + vecInteger, + vecLayout, + vecTableReadable, + type ReadableVecLayout, +} from "./vec-layout.js"; + +export const FTS_SYNC_TRIGGERS_VERSION = 1; +export const VECTOR_PARTITION_VERSION = 2; +export const STORE_SCHEMA_VERSION = VECTOR_PARTITION_VERSION; + +const CURSOR_KEY = "vector_partition_cursor"; + +export function getUserVersion(db: Database): number { + const row = db.prepare(`PRAGMA user_version`).get() as Record | undefined; + const value = row ? Object.values(row)[0] : 0; + return typeof value === "number" ? value : Number(value) || 0; +} + +/** + * Run `step` and stamp `version` in one IMMEDIATE transaction. The + * double-checked read makes concurrent first opens of one database apply the + * step once: busy_timeout serializes the transactions, and the loser sees the + * version the winner stamped. + */ +export function applyVersionedStep(db: Database, version: number, step: () => void): void { + if (getUserVersion(db) >= version) return; + db.exec(`BEGIN IMMEDIATE`); + try { + if (getUserVersion(db) < version) { + step(); + db.exec(`PRAGMA user_version = ${version}`); + } + db.exec(`COMMIT`); + } catch (err) { + db.exec(`ROLLBACK`); + throw err; + } +} + +export type VectorMigrationPhase = "copy" | "verify" | "flip" | "vacuum" | "done"; + +export type VectorMigrationProgress = { + phase: VectorMigrationPhase; + /** Legacy rows processed so far. */ + copied: number; + /** Legacy rows at the start of this process's run. */ + total: number; +}; + +export type VectorMigrationOptions = { + sqliteVecAvailable: boolean; + onProgress?: (progress: VectorMigrationProgress) => void; +}; + +export type VectorMigrationResult = "applied" | "deferred"; + +interface LegacyChunk { + chunkId: number; + size: number; + validity: Uint8Array; + rowids: Uint8Array; + vectors: Uint8Array; +} + +function liveSlots(chunk: { size: number; validity: Uint8Array }): number[] { + const slots: number[] = []; + for (let i = 0; i < chunk.size; i++) { + if ((chunk.validity[i >> 3]! >> (i & 7)) & 1) slots.push(i); + } + return slots; +} + +function splitHashSeq(key: string): { hash: string; seq: number } | null { + const at = key.lastIndexOf("_"); + if (at <= 0) return null; + const seq = Number(key.slice(at + 1)); + return Number.isInteger(seq) ? { hash: key.slice(0, at), seq } : null; +} + +function readCursor(db: Database): number { + const row = db.prepare(`SELECT value FROM store_config WHERE key = ?`).get(CURSOR_KEY) as { value: string } | undefined; + const value = row ? Number(row.value) : Number.NaN; + return Number.isFinite(value) ? value : -1; +} + +function writeCursor(db: Database, chunkId: number): void { + db.prepare(`INSERT INTO store_config (key, value) VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value`) + .run(CURSOR_KEY, String(chunkId)); +} + +/** Writes one vector under every active collection of its hash; returns the rows inserted. */ +class PartitionWriter { + private readonly ids = new Map(); + private readonly collectionsOf; + private readonly insertRow; + private readonly insertVec; + + constructor(private readonly db: Database) { + this.collectionsOf = db.prepare(`SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`); + this.insertRow = db.prepare(`INSERT OR IGNORE INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES (?, ?, ?)`); + this.insertVec = db.prepare(`INSERT INTO ${VEC_TABLE} (rowid, collection_id, embedding) VALUES (?, ?, ?)`); + } + + private idOf(collection: string): number { + let id = this.ids.get(collection); + if (id === undefined) { + id = allocateCollectionId(this.db, collection); + this.ids.set(collection, id); + } + return id; + } + + writeToCollection(hash: string, seq: number, collection: string, embedding: Uint8Array): boolean { + const id = this.idOf(collection); + const inserted = this.insertRow.run(hash, seq, id); + if (inserted.changes === 0) return false; + this.insertVec.run(vecInteger(inserted.lastInsertRowid), vecInteger(id), embedding); + return true; + } + + writeToActiveCollections(hash: string, seq: number, embedding: Uint8Array): number { + let written = 0; + for (const row of this.collectionsOf.all(hash) as { collection: string }[]) { + if (this.writeToCollection(hash, seq, row.collection, embedding)) written++; + } + return written; + } +} + +function copyLegacyChunks( + db: Database, + legacy: ReadableVecLayout, + dimensions: number, + onProgress: (copied: number) => void, +): void { + const writer = new PartitionWriter(db); + const nextChunk = db.prepare(`SELECT chunk_id AS chunkId, size, validity, rowids FROM ${legacy.shadow.chunks} WHERE chunk_id > ? ORDER BY chunk_id LIMIT 1`); + const vectorsOf = db.prepare(`SELECT vectors FROM ${legacy.shadow.vectorChunks} WHERE rowid = ?`); + const keyOf = db.prepare(`SELECT id FROM ${legacy.shadow.rowids} WHERE rowid = ?`); + const hasChunkRow = db.prepare(`SELECT 1 FROM content_vectors WHERE hash = ? AND seq = ?`); + const bytesPerVector = dimensions * 4; + let copied = 0; + + const copyOne = db.transaction((): boolean => { + if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return false; + const chunk = nextChunk.get(readCursor(db)) as Omit | undefined; + if (!chunk) return false; + const blob = (vectorsOf.get(chunk.chunkId) as { vectors: Uint8Array } | undefined)?.vectors; + const rowids = new DataView(chunk.rowids.buffer, chunk.rowids.byteOffset, chunk.rowids.byteLength); + for (const slot of liveSlots(chunk)) { + copied++; + const key = (keyOf.get(rowids.getBigInt64(slot * 8, true)) as { id: string } | undefined)?.id; + const parsed = key === undefined ? null : splitHashSeq(key); + if (!parsed || !blob || blob.byteLength < (slot + 1) * bytesPerVector) continue; + if (!hasChunkRow.get(parsed.hash, parsed.seq)) continue; + const embedding = blob.subarray(slot * bytesPerVector, (slot + 1) * bytesPerVector); + writer.writeToActiveCollections(parsed.hash, parsed.seq, embedding); + } + writeCursor(db, chunk.chunkId); + return true; + }); + + while (copyOne.immediate()) { + onProgress(copied); + } +} + +/** + * Rows the chunk walk can miss: a pre-upgrade `qmd cleanup` repacking the + * legacy table moves live rows into its newest chunk while the walk is past + * it. They are copied by key from the legacy table. + */ +function copyStragglers(db: Database, legacy: ReadableVecLayout): void { + const legacyVector = db.prepare(`SELECT embedding FROM ${legacy.table} WHERE hash_seq = ?`); + db.transaction(() => { + if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return; + const writer = new PartitionWriter(db); + for (const missing of missingPartitionRows(db)) { + const row = legacyVector.get(`${missing.hash}_${missing.seq}`) as { embedding: Uint8Array } | undefined; + if (row) writer.writeToCollection(missing.hash, missing.seq, missing.collection, row.embedding); + } + }).immediate(); +} + +/** + * A content_vectors row with no vector in any partition is a chunk to embed + * again; without the row the pending detector picks the hash up. + */ +export function deleteVectorlessContentVectors(db: Database): number { + return db.prepare(` + DELETE FROM content_vectors + WHERE NOT EXISTS (SELECT 1 FROM ${VEC_ROWS_TABLE} vr WHERE vr.hash = content_vectors.hash AND vr.seq = content_vectors.seq) + `).run().changes; +} + +function flipToPartitioned(db: Database): void { + db.transaction(() => { + if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return; + db.prepare(`DELETE FROM store_config WHERE key = ?`).run(CURSOR_KEY); + db.exec(`PRAGMA user_version = ${VECTOR_PARTITION_VERSION}`); + db.exec(`DROP TABLE IF EXISTS ${LEGACY_VEC_TABLE}`); + }).immediate(); +} + +function ensurePartitionedTable(db: Database, dimensions: number): void { + const layout = vecLayout(db); + const existing = db.prepare(`SELECT sql FROM sqlite_master WHERE type = 'table' AND name = ?`).get(VEC_TABLE) as { sql: string } | undefined; + if (!existing) { + createPartitionedVecTable(db, dimensions); + return; + } + const match = existing.sql.match(/float\[(\d+)\]/); + const existingDims = match?.[1] ? parseInt(match[1], 10) : null; + if (existingDims !== dimensions) { + throw new Error( + `Vector migration found a ${VEC_TABLE} table with ${existingDims ?? "unknown"} dimensions while the legacy ${layout.kind} table holds ${dimensions}d vectors` + ); + } +} + +/** + * The version 2 step. Returns "deferred" when the legacy table cannot be read + * because sqlite-vec is not loaded: the version stays behind and the next open + * with the extension retries, while FTS keeps working. + */ +export function migrateVectorLayout(db: Database, options: VectorMigrationOptions): VectorMigrationResult { + if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return "applied"; + const layout = vecLayout(db); + if (layout.kind !== "legacy") { + applyVersionedStep(db, VECTOR_PARTITION_VERSION, () => {}); + return "applied"; + } + if (!options.sqliteVecAvailable || !vecTableReadable(db, layout)) return "deferred"; + if (!layout.keyedByHashSeq || layout.dimensions === null) { + applyVersionedStep(db, VECTOR_PARTITION_VERSION, () => { + db.exec(`DROP TABLE IF EXISTS ${LEGACY_VEC_TABLE}`); + }); + return "applied"; + } + + const dimensions = layout.dimensions; + const total = (db.prepare(`SELECT COUNT(*) AS c FROM ${layout.shadow.rowids}`).get() as { c: number }).c; + const report = (phase: VectorMigrationPhase, copied: number) => options.onProgress?.({ phase, copied, total }); + + ensurePartitionedTable(db, dimensions); + copyLegacyChunks(db, layout, dimensions, (copied) => report("copy", copied)); + if (getUserVersion(db) < VECTOR_PARTITION_VERSION) { + report("verify", total); + copyStragglers(db, layout); + deleteVectorlessContentVectors(db); + report("flip", total); + flipToPartitioned(db); + report("vacuum", total); + try { + db.exec(`VACUUM`); + } catch (err) { + console.warn(`VACUUM after the vector migration did not run (${err instanceof Error ? err.message : String(err)}); run 'qmd cleanup' to reclaim the space.`); + } + } + report("done", total); + return "applied"; +} + +export type StoreMigrationDeps = VectorMigrationOptions & { + installFtsSyncTriggers: (db: Database) => void; +}; + +export function runStoreMigrations(db: Database, deps: StoreMigrationDeps): void { + applyVersionedStep(db, FTS_SYNC_TRIGGERS_VERSION, () => deps.installFtsSyncTriggers(db)); + if (getUserVersion(db) < VECTOR_PARTITION_VERSION) { + migrateVectorLayout(db, deps); + } +} diff --git a/src/vec-layout.ts b/src/vec-layout.ts index 704d9aae5..f61c0c310 100644 --- a/src/vec-layout.ts +++ b/src/vec-layout.ts @@ -157,6 +157,15 @@ export function deleteCollectionId(db: Database, name: string): void { db.prepare(`DELETE FROM ${VEC_COLLECTION_IDS_TABLE} WHERE name = ?`).run(name); } +/** + * Integer parameter for the vec0 table. better-sqlite3 binds every JS number + * as a double, and vec0 checks the bound type of its partition key and rowid + * instead of applying column affinity, so those parameters go in as bigint. + */ +export function vecInteger(value: number | bigint): bigint { + return BigInt(value); +} + /** * One bound parameter for `json_each(?)`: a scoped search over many * partitions can return more rowids than SQLite allows as separate @@ -165,3 +174,24 @@ export function deleteCollectionId(db: Database, name: string): void { export function rowidList(rowids: readonly number[]): string { return JSON.stringify(rowids); } + +export type MissingPartitionRow = { hash: string; seq: number; collection: string }; + +/** + * Chunks recorded in content_vectors for an active document that have no row + * in that document's collection partition. Each one is a vector to copy from + * another partition (a hash that gained a collection) or, with no partition + * holding it, a chunk to embed again. + */ +export function missingPartitionRows(db: Database, collection?: string): MissingPartitionRow[] { + const filter = collection ? `AND collection = ?` : ``; + return db.prepare(` + SELECT cv.hash, cv.seq, d.collection + FROM content_vectors cv + JOIN (SELECT DISTINCT hash, collection FROM documents WHERE active = 1 ${filter}) d ON d.hash = cv.hash + LEFT JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.name = d.collection + LEFT JOIN ${VEC_ROWS_TABLE} vr ON vr.hash = cv.hash AND vr.seq = cv.seq AND vr.collection_id = ci.id + WHERE vr.id IS NULL + ORDER BY cv.hash, cv.seq, d.collection + `).all(...(collection ? [collection] : [])) as MissingPartitionRow[]; +} diff --git a/test/store-migrations.test.ts b/test/store-migrations.test.ts new file mode 100644 index 000000000..9aa6956d4 --- /dev/null +++ b/test/store-migrations.test.ts @@ -0,0 +1,328 @@ +/** + * The version 2 step copies the legacy `vectors_vec` rows into the + * per-collection partitioned layout, chunk by chunk, resumably, and drops the + * legacy table in the transaction that stamps the version. + */ +import { describe, test, expect, afterEach } from "vitest"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { openDatabase, loadSqliteVec, type Database } from "../src/db.js"; +import { createStore, insertContent, insertDocument, type Store } from "../src/store.js"; +import { + VECTOR_PARTITION_VERSION, + getUserVersion, + migrateVectorLayout, + type VectorMigrationProgress, +} from "../src/store-migrations.js"; +import { + LEGACY_VEC_TABLE, + VEC_ROWS_TABLE, + VEC_TABLE, + resolveCollectionId, + vecInteger, + vecLayout, +} from "../src/vec-layout.js"; + +let store: Store | null = null; +let extra: Database[] = []; +let dir: string | null = null; + +async function openStore(): Promise { + dir = await mkdtemp(join(tmpdir(), "qmd-migration-")); + store = createStore(join(dir, "index.sqlite")); + return store; +} + +function openSecondConnection(s: Store): Database { + const db = openDatabase(s.dbPath); + loadSqliteVec(db); + extra.push(db); + return db; +} + +afterEach(async () => { + for (const db of extra) db.close(); + extra = []; + store?.close(); + store = null; + if (dir) await rm(dir, { recursive: true, force: true }); + dir = null; +}); + +const DIMS = 3; + +function vec(x: number, y: number, z: number): Float32Array { + return new Float32Array([x, y, z]); +} + +function storedVector(db: Database, hash: string, seq: number, collection: string): number[] | undefined { + const id = resolveCollectionId(db, collection); + if (id === undefined) return undefined; + const row = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? AND collection_id = ?`).get(hash, seq, id) as { id: number } | undefined; + if (!row) return undefined; + const stored = db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`).get(vecInteger(row.id)) as { embedding: Uint8Array } | undefined; + if (!stored) return undefined; + return Array.from(new Float32Array(stored.embedding.buffer, stored.embedding.byteOffset, DIMS)); +} + +function partitionCount(db: Database, collection: string): number { + const id = resolveCollectionId(db, collection); + if (id === undefined) return 0; + return (db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE} WHERE collection_id = ?`).get(vecInteger(id)) as { c: number }).c; +} + +function tableNames(db: Database, prefix: string): string[] { + return (db.prepare(`SELECT name FROM sqlite_master WHERE name LIKE ? ORDER BY name`).all(`${prefix}%`) as { name: string }[]).map((r) => r.name); +} + +/** A store still at version 1 with a legacy vec0 table of small chunks. */ +class LegacyFixture { + readonly vectors = new Map(); + + constructor(readonly db: Database, chunkSize = 8) { + db.exec(`PRAGMA user_version = 1`); + db.exec(`CREATE VIRTUAL TABLE ${LEGACY_VEC_TABLE} USING vec0(hash_seq TEXT PRIMARY KEY, embedding float[${DIMS}] distance_metric=cosine, chunk_size=${chunkSize})`); + } + + doc(collection: string, hash: string, active = 1): void { + const now = new Date().toISOString(); + insertContent(this.db, hash, `Document ${hash}`, now); + insertDocument(this.db, collection, `${hash}.md`, hash, hash, now, now); + if (!active) this.db.prepare(`UPDATE documents SET active = 0 WHERE collection = ? AND hash = ?`).run(collection, hash); + } + + chunkRow(hash: string, seq: number, total = 1): void { + this.db.prepare(`INSERT OR IGNORE INTO content_vectors (hash, seq, pos, model, embedded_at, total_chunks) VALUES (?, ?, ?, 'test', ?, ?)`) + .run(hash, seq, seq * 100, new Date().toISOString(), total); + } + + legacyVector(hash: string, seq: number, embedding: Float32Array): void { + this.db.prepare(`INSERT INTO ${LEGACY_VEC_TABLE} (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_${seq}`, embedding); + this.vectors.set(`${hash}_${seq}`, embedding); + } + + /** Document with content_vectors rows and legacy vectors for every chunk. */ + embedded(collection: string, hash: string, chunks: readonly Float32Array[]): void { + this.doc(collection, hash); + chunks.forEach((embedding, seq) => { + this.chunkRow(hash, seq, chunks.length); + this.legacyVector(hash, seq, embedding); + }); + } + + legacyChunkCount(): number { + return (this.db.prepare(`SELECT COUNT(*) AS c FROM ${LEGACY_VEC_TABLE}_chunks`).get() as { c: number }).c; + } +} + +/** + * Two collections sharing one hash, a two-chunk document, an inactive-only + * hash, a legacy row without a content_vectors row, a content_vectors row + * without a vector, and enough filler to span several legacy chunks. + */ +function seedStandardFixture(db: Database): LegacyFixture { + const f = new LegacyFixture(db); + f.embedded("a", "h1", [vec(1, 0, 0), vec(0.9, 0.1, 0)]); + f.embedded("a", "h2", [vec(0, 1, 0)]); + f.embedded("a", "hs", [vec(0, 0, 1)]); + f.doc("b", "hs"); + f.embedded("b", "h3", [vec(0.5, 0.5, 0)]); + f.doc("a", "h4", 0); + f.chunkRow("h4", 0); + f.legacyVector("h4", 0, vec(0.1, 0.2, 0.3)); + f.legacyVector("ghost", 0, vec(0.3, 0.2, 0.1)); + f.doc("a", "h5"); + f.chunkRow("h5", 0); + for (let i = 0; i < 12; i++) { + f.embedded("a", `fill${String(i).padStart(2, "0")}`, [vec(0.2, 0.2, i * 0.05)]); + } + return f; +} + +const STANDARD_A_ROWS = 2 + 1 + 1 + 12; +const STANDARD_B_ROWS = 1 + 1; +/** h1 x2, h2, hs, h3, the inactive h4, the ghost row, and 12 fillers. */ +const STANDARD_LEGACY_ROWS = 2 + 1 + 1 + 1 + 1 + 1 + 12; + +describe("migrateVectorLayout", () => { + test("a fresh store has nothing to copy and reaches the partition version at open", async () => { + const s = await openStore(); + expect(getUserVersion(s.db)).toBeGreaterThanOrEqual(1); + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true })).toBe("applied"); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db).kind).toBe("none"); + }); + + test("copies a hash_seq legacy table into per-collection partitions", async () => { + const s = await openStore(); + const fixture = seedStandardFixture(s.db); + expect(fixture.legacyChunkCount()).toBeGreaterThan(1); + const phases: VectorMigrationProgress[] = []; + + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true, onProgress: (p) => phases.push(p) })).toBe("applied"); + + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db)).toMatchObject({ kind: "partitioned", dimensions: DIMS }); + expect(tableNames(s.db, LEGACY_VEC_TABLE)).toEqual([]); + expect(s.db.prepare(`SELECT value FROM store_config WHERE key = 'vector_partition_cursor'`).get()).toBeFalsy(); + + expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS); + expect((s.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get() as { c: number }).c).toBe(STANDARD_A_ROWS + STANDARD_B_ROWS); + + expect(storedVector(s.db, "h1", 0, "a")).toEqual([1, 0, 0]); + expect(storedVector(s.db, "h1", 1, "a")).toEqual(Array.from(vec(0.9, 0.1, 0))); + expect(storedVector(s.db, "hs", 0, "a")).toEqual([0, 0, 1]); + expect(storedVector(s.db, "hs", 0, "b")).toEqual([0, 0, 1]); + expect(storedVector(s.db, "h3", 0, "b")).toEqual([0.5, 0.5, 0]); + expect(storedVector(s.db, "h3", 0, "a")).toBeUndefined(); + expect(storedVector(s.db, "h4", 0, "a")).toBeUndefined(); + + const chunkRows = (s.db.prepare(`SELECT hash FROM content_vectors ORDER BY hash`).all() as { hash: string }[]).map((r) => r.hash); + expect(chunkRows).not.toContain("h4"); + expect(chunkRows).not.toContain("h5"); + expect(chunkRows).toContain("h1"); + expect(chunkRows).toContain("hs"); + + const nearest = s.db.prepare(`SELECT rowid, distance FROM ${VEC_TABLE} WHERE embedding MATCH ? AND k = 1 AND collection_id = ?`) + .get(vec(0.5, 0.5, 0), vecInteger(resolveCollectionId(s.db, "b")!)) as { rowid: number }; + const nearestRow = s.db.prepare(`SELECT hash FROM ${VEC_ROWS_TABLE} WHERE id = ?`).get(nearest.rowid) as { hash: string }; + expect(nearestRow.hash).toBe("h3"); + + expect(phases.map((p) => p.phase)).toEqual([...phases.filter((p) => p.phase === "copy").map(() => "copy"), "verify", "flip", "vacuum", "done"]); + const lastCopy = phases.filter((p) => p.phase === "copy").at(-1)!; + expect(lastCopy.copied).toBe(lastCopy.total); + expect(lastCopy.total).toBe(STANDARD_LEGACY_ROWS); + }); + + test("drops a legacy table keyed by hash alone and stamps the version", async () => { + const s = await openStore(); + s.db.exec(`PRAGMA user_version = 1`); + s.db.exec(`CREATE VIRTUAL TABLE ${LEGACY_VEC_TABLE} USING vec0(hash TEXT PRIMARY KEY, embedding float[${DIMS}] distance_metric=cosine)`); + s.db.prepare(`INSERT INTO ${LEGACY_VEC_TABLE} (hash, embedding) VALUES ('h1', ?)`).run(vec(1, 0, 0)); + + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true })).toBe("applied"); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db).kind).toBe("none"); + expect(tableNames(s.db, LEGACY_VEC_TABLE)).toEqual([]); + }); + + test("defers without sqlite-vec and leaves the legacy table and version alone", async () => { + const s = await openStore(); + const fixture = seedStandardFixture(s.db); + + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: false })).toBe("deferred"); + expect(getUserVersion(s.db)).toBe(1); + expect(vecLayout(s.db).kind).toBe("legacy"); + expect((s.db.prepare(`SELECT COUNT(*) AS c FROM ${LEGACY_VEC_TABLE}`).get() as { c: number }).c).toBe(fixture.vectors.size); + expect((s.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get() as { c: number }).c).toBe(0); + }); + + test("keeps the legacy dimension", async () => { + const s = await openStore(); + s.db.exec(`PRAGMA user_version = 1`); + s.db.exec(`CREATE VIRTUAL TABLE ${LEGACY_VEC_TABLE} USING vec0(hash_seq TEXT PRIMARY KEY, embedding float[5] distance_metric=cosine)`); + const now = new Date().toISOString(); + insertContent(s.db, "h1", "Document h1", now); + insertDocument(s.db, "a", "h1.md", "h1", "h1", now, now); + s.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES ('h1', 0, 0, 'test', ?)`).run(now); + s.db.prepare(`INSERT INTO ${LEGACY_VEC_TABLE} (hash_seq, embedding) VALUES ('h1_0', ?)`).run(new Float32Array([1, 2, 3, 4, 5])); + + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true })).toBe("applied"); + expect(vecLayout(s.db)).toMatchObject({ kind: "partitioned", dimensions: 5 }); + expect(partitionCount(s.db, "a")).toBe(1); + }); + + test("resumes from the cursor after a failure between chunks", async () => { + const s = await openStore(); + seedStandardFixture(s.db); + let copies = 0; + + expect(() => migrateVectorLayout(s.db, { + sqliteVecAvailable: true, + onProgress: (p) => { + if (p.phase === "copy" && ++copies === 1) throw new Error("killed after the first chunk"); + }, + })).toThrow("killed after the first chunk"); + + expect(getUserVersion(s.db)).toBe(1); + expect(vecLayout(s.db).kind).toBe("legacy"); + const cursor = s.db.prepare(`SELECT value FROM store_config WHERE key = 'vector_partition_cursor'`).get() as { value: string }; + expect(Number(cursor.value)).toBeGreaterThanOrEqual(1); + const partial = (s.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get() as { c: number }).c; + expect(partial).toBeGreaterThan(0); + expect(partial).toBeLessThan(STANDARD_A_ROWS + STANDARD_B_ROWS); + + const resumed: number[] = []; + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true, onProgress: (p) => { if (p.phase === "copy") resumed.push(p.copied); } })).toBe("applied"); + expect(resumed.length).toBeLessThan(copies + 10); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS); + expect((s.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get() as { c: number }).c).toBe(STANDARD_A_ROWS + STANDARD_B_ROWS); + }); + + test("two openers take turns on chunks and both finish with one copy of every row", async () => { + const s = await openStore(); + seedStandardFixture(s.db); + const second = openSecondConnection(s); + let secondResult: string | null = null; + let secondCopies = 0; + + const firstResult = migrateVectorLayout(s.db, { + sqliteVecAvailable: true, + onProgress: (p) => { + if (p.phase === "copy" && secondResult === null) { + secondResult = migrateVectorLayout(second, { + sqliteVecAvailable: true, + onProgress: (q) => { if (q.phase === "copy") secondCopies++; }, + }); + } + }, + }); + + expect(firstResult).toBe("applied"); + expect(secondResult).toBe("applied"); + expect(secondCopies).toBeGreaterThan(0); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db).kind).toBe("partitioned"); + expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS); + expect((s.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get() as { c: number }).c).toBe(STANDARD_A_ROWS + STANDARD_B_ROWS); + }); + + test("the verification pass copies a row that landed in an already-walked chunk", async () => { + const s = await openStore(); + const fixture = seedStandardFixture(s.db); + let injected = false; + + expect(migrateVectorLayout(s.db, { + sqliteVecAvailable: true, + onProgress: (p) => { + if (p.phase === "copy" && p.copied === p.total && !injected) { + injected = true; + fixture.embedded("b", "late", [vec(0.7, 0.7, 0.1)]); + } + }, + })).toBe("applied"); + + expect(injected).toBe(true); + expect(storedVector(s.db, "late", 0, "b")).toEqual(Array.from(vec(0.7, 0.7, 0.1))); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS + 1); + }); + + test("the flip drops the legacy table and its shadow tables with the version stamp", async () => { + const s = await openStore(); + seedStandardFixture(s.db); + expect(tableNames(s.db, LEGACY_VEC_TABLE).length).toBeGreaterThan(1); + + migrateVectorLayout(s.db, { sqliteVecAvailable: true }); + + expect(tableNames(s.db, LEGACY_VEC_TABLE)).toEqual([]); + expect(tableNames(s.db, VEC_TABLE)).toContain(`${VEC_TABLE}_chunks`); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true })).toBe("applied"); + expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); + }); +}); From a1f55cef814c0b25f6e40433cab530b6ec483a1e Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 24 Sep 2026 19:32:48 -0500 Subject: [PATCH 30/82] feat(store): partition the vector index by collection Vector search, embedding writes, cleanup and the collection operations now use the layout vec-layout.ts describes: one vec0 table partitioned by an integer collection id with one row per (chunk, active collection), a `vector_rows` mapping from vec0 rowid to (hash, seq, collection_id), and the version 2 migration wired into store open so every process copies the legacy `vectors_vec` table before it can read vectors. A collection-scoped `searchVec` is one KNN per collection id with `k = limit * 3` (capped at sqlite-vec's 4096), merged by distance, so a small collection costs its own rows instead of a pre-filter over the whole index and is never crowded out by a large one (#775, #791, #803). The unscoped search is one KNN over all partitions with the same k. Step two joins the returned rowids through `json_each` in a single bound parameter, because thirteen partitions at the k cap exceed SQLite's separate-parameter limit. A scope naming an unknown collection skips it; an all-unknown scope is empty. The exact-scan, capped-ANN and post-filter paths are gone, as is the per-collection recursion; `mergeSearchResultsByScore` stays for FTS. A metadata filter restricts each KNN to the vector rows of documents it admits, passed as `rowid IN (SELECT value FROM json_each(?))`, which vec0 0.1.9 applies inside the scan alongside the partition key. The result is an exact top-k of the eligible rows at any eligible-set size, so the 20,000-row exact-scan cutoff and the global over-fetch above it go away, and with them the false empty a large eligible set got when more than the over-fetch of closer rows were ineligible. The filter is re-applied on the document join, since one hash can back a matching and a non-matching document in the same collection. On the live index the restriction costs nothing measurable: a stars KNN (794k rows) restricted to half its rows took 1.04 s against 1.16 s unrestricted. `insertEmbedding` writes the content_vectors row and one partition row per active collection of the hash in one transaction, replacing an existing row under the same rowid since vec0 ignores OR REPLACE. `cleanupOrphanedVectors` removes partition rows whose (hash, collection) has no active document, so a stale row in one collection cannot steal k slots there while the hash stays live elsewhere. `clearAllEmbeddings(collection)` clears the collection's partition rows for hashes exclusive to it and keeps shared ones in every partition, `removeCollection` deletes the partition and its id, `renameCollection` updates the id's name, and `removeIncompleteEmbeddings` drops every partition row of the hash. Doctor's stored-vector check and the legacy-fingerprint adoption sample resolve the vector through `vector_rows`. `ensureVecTable` keeps the dimension-mismatch check on the partitioned table and runs the copy itself if it meets a legacy table, which covers an older build having recreated one on an already-migrated index. The metadata search suite gains the over-20,000-eligible case, which returns 0 of 5 documents on the unpartitioned code, and a same-collection shared-hash case. The store-level tests move to `insertEmbedding` fixtures, the MCP fixture seeds the legacy table and runs the migration, and the collection-filter suite gains the scoped-search semantics tests (shared hashes named per collection, inactive rows, unknown and empty scopes, a limit above the k cap, one-row-per-file collapse, per-partition k, rename, remove and scoped clear). No reader needs a retry for the moment the legacy table is dropped: the migration runs inside `createStore`, so no process using this code reads vectors before its own open has finished the copy. (cherry picked from commit c8cbc4fed5a7499bb7a8856480565a18a5ba985e) --- src/cli/embed-lock.ts | 5 +- src/cli/qmd.ts | 10 +- src/store-migrations.ts | 57 +-- src/store.ts | 569 +++++++++++++-------------- src/vec-layout.ts | 81 ++++ test/eval.test.ts | 16 +- test/mcp.test.ts | 7 + test/metadata-search.test.ts | 43 ++ test/multi-collection-filter.test.ts | 7 +- test/store.helpers.unit.test.ts | 8 +- test/store.test.ts | 507 +++++++++++++++++++++--- 11 files changed, 883 insertions(+), 427 deletions(-) diff --git a/src/cli/embed-lock.ts b/src/cli/embed-lock.ts index ac95f5568..e05ffac2f 100644 --- a/src/cli/embed-lock.ts +++ b/src/cli/embed-lock.ts @@ -1,8 +1,9 @@ /** * Process-level exclusive lock for `qmd embed`. * - * Concurrent embed runs against the same index can race on vectors_vec - * (UNIQUE constraint on hash_seq). This lockfile keeps a second process from + * Concurrent embed runs against the same index can race on vector_rows + * (UNIQUE constraint on hash, seq, collection_id). This lockfile keeps a + * second process from * starting while another embed holds the lock. Stale files left by crashed * processes are recovered via PID identity checks (same spirit as mcp-pid.ts). */ diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index eedd2aa27..db51bc794 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -103,6 +103,7 @@ import { import { formatMetadataKeySummaries, formatMetadataOverview } from "../metadata-format.js"; import type { DocumentMetadata } from "../metadata.js"; import { parseMetadataFilter, parseMetadataMatch, type MetadataFilter, type MetadataMatch } from "../metadata-filter.js"; +import { hasVectorIndex, storedEmbedding } from "../vec-layout.js"; import { disposeDefaultLlamaCpp, getDefaultLlamaCpp, setDefaultLlamaCpp, LlamaCpp, withLLMSession, pullModels, DEFAULT_MODEL_CACHE_DIR, resolveEmbedModel, resolveGenerateModel, resolveRerankModel, resolveModels, inspectGgufFile, isDarwinMetalMitigationActive } from "../llm.js"; import { formatSearchResults, @@ -2379,7 +2380,7 @@ async function vectorIndex( const storeInstance = getStore(); const db = storeInstance.db; - // Exclusive process lock — concurrent embeds race on vectors_vec (#825) + // Exclusive process lock — concurrent embeds race on vector_rows (#825) const embedLock = tryAcquireEmbedLock(embedLockPathForDb(getDbPath())); if (!embedLock) { console.log(EMBED_LOCK_BUSY_MESSAGE); @@ -4225,8 +4226,7 @@ export async function checkEmbeddingVectorSamples(db: Database, model: string, f return { ok: true, details: "no active documents indexed" }; } - const vecTableExists = db.prepare(`SELECT 1 FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get(); - if (!vecTableExists) { + if (!hasVectorIndex(db)) { return { ok: false, details: "no vector table to test; please run qmd embed again" }; } @@ -4258,13 +4258,13 @@ export async function checkEmbeddingVectorSamples(db: Database, model: string, f continue; } - const stored = db.prepare(`SELECT embedding FROM vectors_vec WHERE hash_seq = ?`).get(hashSeq) as { embedding: Uint8Array } | undefined; + const stored = storedEmbedding(db, sample.hash, sample.seq); if (!stored) { mismatches.push(`${shortHashSeq(hashSeq)}: stored vector missing`); continue; } - const distance = cosineDistance(result.embedding, decodeStoredEmbedding(stored.embedding)); + const distance = cosineDistance(result.embedding, decodeStoredEmbedding(stored)); if (distance > threshold) { mismatches.push(`${shortHashSeq(hashSeq)}: stored vector distance ${distance.toFixed(6)}`); } diff --git a/src/store-migrations.ts b/src/store-migrations.ts index 5bfbedb64..ead3635d1 100644 --- a/src/store-migrations.ts +++ b/src/store-migrations.ts @@ -17,12 +17,11 @@ import type { Database } from "./db.js"; import { LEGACY_VEC_TABLE, + PartitionWriter, VEC_ROWS_TABLE, VEC_TABLE, - allocateCollectionId, createPartitionedVecTable, missingPartitionRows, - vecInteger, vecLayout, vecTableReadable, type ReadableVecLayout, @@ -112,45 +111,6 @@ function writeCursor(db: Database, chunkId: number): void { .run(CURSOR_KEY, String(chunkId)); } -/** Writes one vector under every active collection of its hash; returns the rows inserted. */ -class PartitionWriter { - private readonly ids = new Map(); - private readonly collectionsOf; - private readonly insertRow; - private readonly insertVec; - - constructor(private readonly db: Database) { - this.collectionsOf = db.prepare(`SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`); - this.insertRow = db.prepare(`INSERT OR IGNORE INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES (?, ?, ?)`); - this.insertVec = db.prepare(`INSERT INTO ${VEC_TABLE} (rowid, collection_id, embedding) VALUES (?, ?, ?)`); - } - - private idOf(collection: string): number { - let id = this.ids.get(collection); - if (id === undefined) { - id = allocateCollectionId(this.db, collection); - this.ids.set(collection, id); - } - return id; - } - - writeToCollection(hash: string, seq: number, collection: string, embedding: Uint8Array): boolean { - const id = this.idOf(collection); - const inserted = this.insertRow.run(hash, seq, id); - if (inserted.changes === 0) return false; - this.insertVec.run(vecInteger(inserted.lastInsertRowid), vecInteger(id), embedding); - return true; - } - - writeToActiveCollections(hash: string, seq: number, embedding: Uint8Array): number { - let written = 0; - for (const row of this.collectionsOf.all(hash) as { collection: string }[]) { - if (this.writeToCollection(hash, seq, row.collection, embedding)) written++; - } - return written; - } -} - function copyLegacyChunks( db: Database, legacy: ReadableVecLayout, @@ -166,7 +126,7 @@ function copyLegacyChunks( let copied = 0; const copyOne = db.transaction((): boolean => { - if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return false; + if (vecLayout(db).kind !== "legacy") return false; const chunk = nextChunk.get(readCursor(db)) as Omit | undefined; if (!chunk) return false; const blob = (vectorsOf.get(chunk.chunkId) as { vectors: Uint8Array } | undefined)?.vectors; @@ -197,7 +157,7 @@ function copyLegacyChunks( function copyStragglers(db: Database, legacy: ReadableVecLayout): void { const legacyVector = db.prepare(`SELECT embedding FROM ${legacy.table} WHERE hash_seq = ?`); db.transaction(() => { - if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return; + if (vecLayout(db).kind !== "legacy") return; const writer = new PartitionWriter(db); for (const missing of missingPartitionRows(db)) { const row = legacyVector.get(`${missing.hash}_${missing.seq}`) as { embedding: Uint8Array } | undefined; @@ -219,9 +179,9 @@ export function deleteVectorlessContentVectors(db: Database): number { function flipToPartitioned(db: Database): void { db.transaction(() => { - if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return; + if (vecLayout(db).kind !== "legacy") return; db.prepare(`DELETE FROM store_config WHERE key = ?`).run(CURSOR_KEY); - db.exec(`PRAGMA user_version = ${VECTOR_PARTITION_VERSION}`); + db.exec(`PRAGMA user_version = ${Math.max(getUserVersion(db), VECTOR_PARTITION_VERSION)}`); db.exec(`DROP TABLE IF EXISTS ${LEGACY_VEC_TABLE}`); }).immediate(); } @@ -245,10 +205,11 @@ function ensurePartitionedTable(db: Database, dimensions: number): void { /** * The version 2 step. Returns "deferred" when the legacy table cannot be read * because sqlite-vec is not loaded: the version stays behind and the next open - * with the extension retries, while FTS keeps working. + * with the extension retries, while FTS keeps working. The step keys on the + * legacy table rather than the version so that a legacy table an older build + * created on an already-stamped database is copied too. */ export function migrateVectorLayout(db: Database, options: VectorMigrationOptions): VectorMigrationResult { - if (getUserVersion(db) >= VECTOR_PARTITION_VERSION) return "applied"; const layout = vecLayout(db); if (layout.kind !== "legacy") { applyVersionedStep(db, VECTOR_PARTITION_VERSION, () => {}); @@ -268,7 +229,7 @@ export function migrateVectorLayout(db: Database, options: VectorMigrationOption ensurePartitionedTable(db, dimensions); copyLegacyChunks(db, layout, dimensions, (copied) => report("copy", copied)); - if (getUserVersion(db) < VECTOR_PARTITION_VERSION) { + if (vecLayout(db).kind === "legacy") { report("verify", total); copyStragglers(db, layout); deleteVectorlessContentVectors(db); diff --git a/src/store.ts b/src/store.ts index 3617115ac..2d0cb06e1 100644 --- a/src/store.ts +++ b/src/store.ts @@ -12,8 +12,34 @@ */ import { openDatabase, loadSqliteVec } from "./db.js"; -import { createVectorMetadataTables } from "./vec-layout.js"; -import type { Database } from "./db.js"; +import { + VEC_COLLECTION_IDS_TABLE, + VEC_ROWS_TABLE, + VEC_TABLE, + allocateCollectionId, + createPartitionedVecTable, + createVectorMetadataTables, + deleteCollectionId, + deletePartitionRows, + hasVectorIndex, + partitionRowKey, + renameCollectionId, + resolveCollectionId, + resolveCollectionIds, + rowidList, + upsertPartitionVector, + vecInteger, + vecLayout, + vecTableReadable, +} from "./vec-layout.js"; +import { + FTS_SYNC_TRIGGERS_VERSION, + getUserVersion, + migrateVectorLayout, + runStoreMigrations, + type VectorMigrationProgress, +} from "./store-migrations.js"; +import type { Database, SQLiteValue } from "./db.js"; import picomatch from "picomatch"; import { createHash } from "crypto"; import { readFileSync, realpathSync, statSync, mkdirSync } from "node:fs"; @@ -890,10 +916,6 @@ const CJK_CHAR_PATTERN = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\ const CJK_RUN_PATTERN = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]+/gu; const FTS_CJK_NORMALIZED_VERSION = "1"; -// Bump when any FTS sync trigger body in applyFtsSyncTriggers changes, so the -// new definition is reapplied to existing databases on next open. -const STORE_SCHEMA_VERSION = 1; - /** * FTS5's unicode61 tokenizer does not segment CJK text into searchable words. * Normalize CJK runs by spacing every character so exact CJK queries can be @@ -920,12 +942,6 @@ function sanitizeFTS5Phrase(phrase: string): string { .join(' '); } -function getUserVersion(db: Database): number { - const row = db.prepare(`PRAGMA user_version`).get() as Record | undefined; - const value = row ? Object.values(row)[0] : 0; - return typeof value === "number" ? value : Number(value) || 0; -} - // FTS sync triggers keep documents_fts current for callers that write directly // to documents (production indexing rebuilds FTS in TypeScript to normalize CJK // first). The bodies use DROP+CREATE rather than CREATE IF NOT EXISTS so a @@ -933,9 +949,11 @@ function getUserVersion(db: Database): number { // autocommit statements, so concurrent opens of one database interleave across // connections (A drops, B drops, A creates, B creates -> "trigger already // exists"); busy_timeout serializes individual statements but not the pair. -// Gate the work behind PRAGMA user_version and apply it inside one IMMEDIATE -// transaction: the DROP+CREATE pair is atomic across connections, and a -// double-checked read skips it once any process has stamped the version. +// runStoreMigrations gates the work behind PRAGMA user_version and applies it +// inside one IMMEDIATE transaction: the DROP+CREATE pair is atomic across +// connections, and a double-checked read skips it once any process has +// stamped the version. Bump FTS_SYNC_TRIGGERS_VERSION when a trigger body +// changes so existing databases reinstall it on the next open. function installFtsSyncTriggers(db: Database): void { db.exec(`DROP TRIGGER IF EXISTS documents_ai`); db.exec(` @@ -978,19 +996,25 @@ function installFtsSyncTriggers(db: Database): void { `); } -function applyFtsSyncTriggers(db: Database): void { - if (getUserVersion(db) >= STORE_SCHEMA_VERSION) return; - db.exec(`BEGIN IMMEDIATE`); - try { - if (getUserVersion(db) < STORE_SCHEMA_VERSION) { - installFtsSyncTriggers(db); - db.exec(`PRAGMA user_version = ${STORE_SCHEMA_VERSION}`); +/** Progress of the one-time vector layout upgrade, on stderr. */ +function vectorMigrationReporter(): (progress: VectorMigrationProgress) => void { + let startedAt: number | null = null; + const tty = Boolean(process.stderr.isTTY); + return ({ phase, copied, total }) => { + if (startedAt === null) { + startedAt = Date.now(); + process.stderr.write(`Moving ${total} vectors to the per-collection index layout (one-time upgrade)...\n`); } - db.exec(`COMMIT`); - } catch (err) { - db.exec(`ROLLBACK`); - throw err; - } + if (phase === "copy") { + if (tty) process.stderr.write(`\r copied ${copied}/${total}`); + return; + } + if (tty) process.stderr.write("\r"); + if (phase === "verify") process.stderr.write(` copied ${copied}/${total}; verifying...\n`); + else if (phase === "flip") process.stderr.write(` dropping the old vector table...\n`); + else if (phase === "vacuum") process.stderr.write(` vacuuming the index (minutes on a large index)...\n`); + else process.stderr.write(` vector index upgraded in ${Math.round((Date.now() - startedAt) / 1000)}s\n`); + }; } /** @@ -1067,7 +1091,7 @@ function recreateDocumentsFts(db: Database): void { } // Missing-table create and legacy-schema repair share one IMMEDIATE -// transaction with a double-checked read, matching applyFtsSyncTriggers: +// transaction with a double-checked read, matching applyVersionedStep: // the DROP+CREATE (or first CREATE) is atomic across connections, and // losers skip once any process has published the current table. function ensureDocumentsFtsSchema(db: Database): void { @@ -1077,10 +1101,10 @@ function ensureDocumentsFtsSchema(db: Database): void { if (!documentsFtsSchemaIsCurrent(db)) { if (documentsFtsExists(db)) { recreateDocumentsFts(db); - // recreateDocumentsFts dropped the sync triggers. applyFtsSyncTriggers + // recreateDocumentsFts dropped the sync triggers. runStoreMigrations // only reinstalls them when user_version is stale, so a DB that already // has the current user_version would otherwise be left untriggered. - if (getUserVersion(db) >= STORE_SCHEMA_VERSION) { + if (getUserVersion(db) >= FTS_SYNC_TRIGGERS_VERSION) { installFtsSyncTriggers(db); } } else { @@ -1322,7 +1346,11 @@ function initializeDatabase(db: Database): void { // Do not CREATE VIRTUAL TABLE here as an autocommit statement: FTS5 // IF NOT EXISTS races under WAL (see createDocumentsFtsTable). ensureDocumentsFtsSchema(db); - applyFtsSyncTriggers(db); + runStoreMigrations(db, { + installFtsSyncTriggers, + sqliteVecAvailable: _sqliteVecAvailable === true, + onProgress: vectorMigrationReporter(), + }); rebuildFTSForCjkNormalization(db); } @@ -1512,22 +1540,23 @@ function ensureVecTableInternal(db: Database, dimensions: number): void { _sqliteVecUnavailableReason ?? "vector operations require a SQLite build with extension loading support" ); } - const tableInfo = db.prepare(`SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get() as { sql: string } | null; - if (tableInfo) { - const match = tableInfo.sql.match(/float\[(\d+)\]/); - const hasHashSeq = tableInfo.sql.includes('hash_seq'); - const hasCosine = tableInfo.sql.includes('distance_metric=cosine'); - const existingDims = match?.[1] ? parseInt(match[1], 10) : null; - if (existingDims === dimensions && hasHashSeq && hasCosine) return; - if (existingDims !== null && existingDims !== dimensions) { + let layout = vecLayout(db); + if (layout.kind === "legacy") { + migrateVectorLayout(db, { sqliteVecAvailable: true, onProgress: vectorMigrationReporter() }); + layout = vecLayout(db); + } + if (layout.kind === "partitioned") { + if (layout.dimensions === dimensions) return; + if (layout.dimensions !== null) { throw new Error( - `Embedding dimension mismatch: existing vectors are ${existingDims}d but the current model produces ${dimensions}d. ` + + `Embedding dimension mismatch: existing vectors are ${layout.dimensions}d but the current model produces ${dimensions}d. ` + `Run 'qmd embed -f' to re-embed with the new model.` ); } - db.exec("DROP TABLE IF EXISTS vectors_vec"); + db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); + db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); } - db.exec(`CREATE VIRTUAL TABLE vectors_vec USING vec0(hash_seq TEXT PRIMARY KEY, embedding float[${dimensions}] distance_metric=cosine)`); + createPartitionedVecTable(db, dimensions); } // ============================================================================= @@ -2953,9 +2982,8 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store: Store, model: return { checked: false, adopted: 0, reason: `${legacyCount} legacy docs have no active sample` }; } - const tableExists = db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get(); - if (!tableExists) { - return { checked: false, adopted: 0, reason: "vectors_vec table is missing" }; + if (!hasVectorIndex(db)) { + return { checked: false, adopted: 0, reason: "vector index is missing" }; } const expectedHashSeq = `${sample.hash}_${sample.seq}`; @@ -2984,18 +3012,20 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store: Store, model: } const nearest = db.prepare(` - SELECT hash_seq, distance - FROM vectors_vec + SELECT rowid, distance + FROM ${VEC_TABLE} WHERE embedding MATCH ? AND k = 1 - `).get(new Float32Array(result.embedding)) as { hash_seq: string; distance: number } | undefined; + `).get(new Float32Array(result.embedding)) as { rowid: number; distance: number } | undefined; + const nearestKey = nearest ? partitionRowKey(db, nearest.rowid) : undefined; - if (!nearest) { + if (!nearest || !nearestKey) { return { checked: true, adopted: 0, reason: "legacy sample vector not found" }; } const threshold = 0.0001; - if (nearest.hash_seq !== expectedHashSeq || nearest.distance > threshold) { - return { checked: true, adopted: 0, reason: `legacy sample differs from current fingerprint (nearest ${nearest.hash_seq}, distance ${nearest.distance.toFixed(6)})` }; + const nearestHashSeq = `${nearestKey.hash}_${nearestKey.seq}`; + if (nearestHashSeq !== expectedHashSeq || nearest.distance > threshold) { + return { checked: true, adopted: 0, reason: `legacy sample differs from current fingerprint (nearest ${nearestHashSeq}, distance ${nearest.distance.toFixed(6)})` }; } const update = withLazyContentVectorMigration(db, () => db.prepare(`UPDATE content_vectors SET embed_fingerprint = ? WHERE model = ? AND embed_fingerprint = ''`).run(fingerprint, model)); @@ -3125,68 +3155,51 @@ export function countOrphanedVectors(db: Database): number { } /** - * Remove orphaned vector embeddings that are not referenced by any active document. - * Returns the number of orphaned embedding chunks deleted. + * Remove vector rows whose (hash, collection) no active document references, + * and the content_vectors rows of hashes no active document references at + * all. Returns the number of orphaned chunks deleted. */ export function cleanupOrphanedVectors(db: Database): number { // sqlite-vec may not be loaded (e.g. Bun's bun:sqlite lacks loadExtension). - // The vectors_vec virtual table can appear in sqlite_master from a prior - // session, but querying it without the vec0 module loaded will crash (#380). + // The vec0 table can appear in sqlite_master from a prior session, but + // querying it without the module loaded would crash (#380). if (!isSqliteVecAvailable()) { return 0; } - - // The schema entry can exist even when sqlite-vec itself is unavailable - // (for example when reopening a DB without vec0 loaded). In that case, - // touching the virtual table throws "no such module: vec0" and cleanup - // should degrade gracefully like the rest of the vector features. - try { - db.prepare(`SELECT 1 FROM vectors_vec LIMIT 0`).get(); - } catch { + const layout = vecLayout(db); + if (layout.kind !== "partitioned" || !vecTableReadable(db, layout)) { return 0; } return withLazyContentVectorMigration(db, () => { - // Count and both DELETEs share one transaction. An interruption between the - // two DELETEs (crash, SQLITE_BUSY) desyncs the tables: vectors_vec loses - // the rows while content_vectors still records the chunks as embedded. - // These rows are orphaned (no active document), so live vector search — - // which post-filters on documents.active = 1 — is unaffected right away. - // The failure is latent: if that content hash is later reactivated (qmd is - // content-addressable, so the same content returning revives the hash), the - // stale content_vectors rows make getHashesNeedingEmbedding treat it as - // already embedded, so qmd embed skips it and the document is silently - // unsearchable by vector with no orphan left to clean up. Keeping the count - // inside the same transaction also makes the returned number match the rows - // the DELETEs actually remove if another connection mutates documents - // concurrently. Run it BEGIN IMMEDIATE: the count reads before the DELETEs - // write, and upgrading a deferred read snapshot under a concurrent WAL - // writer fails with SQLITE_BUSY_SNAPSHOT instead of honoring the busy + // The count and both DELETEs share one transaction: an interruption + // between them (crash, SQLITE_BUSY) would leave content_vectors claiming + // chunks the index no longer holds, and a revived hash would then be + // skipped by getHashesNeedingEmbedding and stay unsearchable. Run it + // BEGIN IMMEDIATE: upgrading a deferred read snapshot under a concurrent + // WAL writer fails with SQLITE_BUSY_SNAPSHOT instead of honoring the busy // timeout. Nested callers still get a savepoint. const cleanup = db.transaction(() => { - const orphaned = (db.prepare(ORPHANED_VECTOR_COUNT_SQL).get() as { c: number }).c; - if (orphaned === 0) { + const orphanRows = db.prepare(` + SELECT vr.id FROM ${VEC_ROWS_TABLE} vr + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + WHERE NOT EXISTS ( + SELECT 1 FROM documents d WHERE d.hash = vr.hash AND d.collection = ci.name AND d.active = 1 + ) + `).all() as { id: number }[]; + const orphanedChunks = (db.prepare(ORPHANED_VECTOR_COUNT_SQL).get() as { c: number }).c; + if (orphanRows.length === 0 && orphanedChunks === 0) { return 0; } - // Delete from vectors_vec first - db.exec(` - DELETE FROM vectors_vec WHERE hash_seq IN ( - SELECT cv.hash || '_' || cv.seq FROM content_vectors cv - WHERE NOT EXISTS ( - SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1 - ) - ) - `); - - // Delete from content_vectors + deletePartitionRows(db, orphanRows.map((row) => row.id)); db.exec(` DELETE FROM content_vectors WHERE hash NOT IN ( SELECT hash FROM documents WHERE active = 1 ) `); - return orphaned; + return Math.max(orphanRows.length, orphanedChunks); }); return cleanup.immediate(); @@ -4036,11 +4049,32 @@ export function listCollections(db: Database): { name: string; pwd: string; glob return result; } +/** + * Drop a collection's vector partition and its id. With the vec0 table + * present but sqlite-vec not loaded, the rows stay for cleanupOrphanedVectors + * to remove once the extension loads, so the mapping never runs ahead of the + * index. + */ +function deleteVectorPartition(db: Database, collectionName: string): void { + const collectionId = resolveCollectionId(db, collectionName); + if (collectionId === undefined) return; + const layout = vecLayout(db); + if (layout.kind === "legacy") return; + if (layout.kind === "partitioned") { + if (!isSqliteVecAvailable()) return; + db.prepare(`DELETE FROM ${VEC_TABLE} WHERE collection_id = ?`).run(vecInteger(collectionId)); + } + db.prepare(`DELETE FROM ${VEC_ROWS_TABLE} WHERE collection_id = ?`).run(collectionId); + deleteCollectionId(db, collectionName); +} + /** * Remove a collection and clean up its documents. * Uses collections.ts to remove from YAML config and cleans up database. */ export function removeCollection(db: Database, collectionName: string): { deletedDocs: number; cleanedHashes: number } { + deleteVectorPartition(db, collectionName); + // Delete documents from database const docResult = db.prepare(`DELETE FROM documents WHERE collection = ?`).run(collectionName); db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(collectionName); @@ -4072,6 +4106,7 @@ export function renameCollection(db: Database, oldName: string, newName: string) // under the new name. Rows already under it belong to no collection. db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(newName); db.prepare(`UPDATE file_sync_state SET collection = ? WHERE collection = ?`).run(newName, oldName); + renameCollectionId(db, oldName, newName); // Rename in store_collections renameStoreCollection(db, oldName, newName); @@ -4550,153 +4585,127 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle /** sqlite-vec rejects k above this in MATCH queries (v0.1.9). */ const SQLITE_VEC_MAX_K = 4096; -/** - * Max filter-eligible vectors for an exact cosine scan. Above this we fall - * back to global ANN with a capped over-fetch. Exact scan avoids the - * post-filter starvation of small eligible sets — originally small - * collections (#791, #803), now also selective metadata filters; ANN remains - * for very large eligible sets where a full scan would be expensive. - */ -const FILTERED_VEC_EXACT_SCAN_MAX = 20_000; +interface VecMatch { + rowid: number; + distance: number; +} -const VEC_HASH_SEQ_IN_CHUNK = 400; +/** One KNN scan target: a partition (or the whole table) and the rows a filter admits in it. */ +interface VecScanTarget { + collectionId?: number; + eligibleRowids?: readonly number[]; +} /** - * Exact cosine-distance scan over a known set of hash_seq keys. - * Uses vec_distance_cosine with chunked IN lists (no JOIN with vectors_vec). + * Prepares an exact top-k cosine scan of the vector table, run once per + * target. vec0 evaluates the partition equality inside the scan, so a small + * collection costs its own rows rather than the whole index and is never + * crowded out by a larger one (#775, #791, #803). A `rowid IN` restriction is + * applied inside the scan the same way, so a selective metadata filter gets an + * exact top-k of its own rows instead of whatever survives a post-filter of a + * larger top-k. */ -function exactVecScanByHashSeq( - db: Database, - embedding: number[], - hashSeqs: string[], - limit: number, -): { hash_seq: string; distance: number }[] { - if (hashSeqs.length === 0 || limit <= 0) return []; - - const queryVec = new Float32Array(embedding); - // Over-fetch a bit so multi-chunk docs can still yield `limit` unique files. - const fetchLimit = Math.max(limit * 3, limit); - const scored: { hash_seq: string; distance: number }[] = []; - - for (let i = 0; i < hashSeqs.length; i += VEC_HASH_SEQ_IN_CHUNK) { - const chunk = hashSeqs.slice(i, i + VEC_HASH_SEQ_IN_CHUNK); - const placeholders = chunk.map(() => "?").join(","); - const rows = db.prepare(` - SELECT hash_seq, vec_distance_cosine(embedding, ?) AS distance - FROM vectors_vec - WHERE hash_seq IN (${placeholders}) - `).all(queryVec, ...chunk) as { hash_seq: string; distance: number }[]; - scored.push(...rows); - } - - scored.sort((a, b) => a.distance - b.distance); - return scored.slice(0, fetchLimit); +function knnVecScanner(db: Database, partitioned: boolean, restricted: boolean): (embedding: Float32Array, k: number, target: VecScanTarget) => VecMatch[] { + const conditions = ["embedding MATCH ?", "k = ?"]; + if (partitioned) conditions.push("collection_id = ?"); + if (restricted) conditions.push("rowid IN (SELECT value FROM json_each(?))"); + const statement = db.prepare(` + SELECT rowid, distance + FROM ${VEC_TABLE} + WHERE ${conditions.join(" AND ")} + `); + return (embedding, k, target) => { + const params: SQLiteValue[] = [embedding, Math.max(1, Math.min(SQLITE_VEC_MAX_K, k))]; + if (partitioned) params.push(vecInteger(target.collectionId ?? 0)); + if (restricted) params.push(rowidList(target.eligibleRowids ?? [])); + return statement.all(...params) as VecMatch[]; + }; } -function annVecScan( - db: Database, - embedding: number[], - k: number, -): { hash_seq: string; distance: number }[] { - const vecK = Math.max(1, Math.min(SQLITE_VEC_MAX_K, k)); - return db.prepare(` - SELECT hash_seq, distance - FROM vectors_vec - WHERE embedding MATCH ? AND k = ? - `).all(new Float32Array(embedding), vecK) as { hash_seq: string; distance: number }[]; +/** + * Vector rows of active documents a metadata filter admits, keyed by + * collection id. A row is eligible when any document of its collection that + * holds the row's content passes the filter. + */ +function metadataEligibleVectorRows(db: Database, filter: MetadataFilter, collectionIds?: readonly number[]): Map { + const compiled = compileMetadataFilter(filter, "d"); + const params: SQLiteValue[] = [...compiled.params]; + let scope = ""; + if (collectionIds) { + scope = ` AND vr.collection_id IN (SELECT value FROM json_each(?))`; + params.push(JSON.stringify(collectionIds)); + } + const rows = db.prepare(` + SELECT DISTINCT vr.id AS rowid, vr.collection_id AS collectionId + FROM documents d + JOIN document_metadata dm ON dm.document_id = d.id + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.name = d.collection + JOIN ${VEC_ROWS_TABLE} vr ON vr.hash = d.hash AND vr.collection_id = ci.id + WHERE d.active = 1 + AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} + AND dm.extraction_error IS NULL + AND ${compiled.sql}${scope} + `).all(...params) as { rowid: number; collectionId: number }[]; + const byCollection = new Map(); + for (const row of rows) { + const list = byCollection.get(row.collectionId); + if (list) list.push(row.rowid); + else byCollection.set(row.collectionId, [row.rowid]); + } + return byCollection; } export async function searchVec(db: Database, query: string, model: string, limit: number = 20, collectionName?: string | readonly string[], session?: ILLMSession, precomputedEmbedding?: number[], llm?: LlamaCpp, filter?: MetadataFilter): Promise { - const tableExists = db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get(); - if (!tableExists) return []; + if (!hasVectorIndex(db)) return []; const embedding = precomputedEmbedding ?? await getEmbedding(query, model, true, session, llm); if (!embedding) return []; const names = scopedCollectionNames(collectionName); - if (names && names.length > 1) { - const lists = await Promise.all( - names.map(name => searchVec(db, query, model, limit, name, session, embedding, llm, filter)), - ); - return mergeSearchResultsByScore(lists, limit); + let collectionIds: number[] | undefined; + if (names) { + collectionIds = Array.from(resolveCollectionIds(db, names).values()); + if (collectionIds.length === 0) return []; } - const collectionFilter = names?.[0]; + const eligible = filter ? metadataEligibleVectorRows(db, filter, collectionIds) : undefined; // IMPORTANT: We use a two-step query approach here because sqlite-vec virtual tables // hang indefinitely when combined with JOINs in the same query. Do NOT try to // "optimize" this by combining into a single query with JOINs - it will break. // See: https://github.com/tobi/qmd/pull/23 - // Step 1: Get vector matches from sqlite-vec (no JOINs allowed). - // - // Collection and metadata filters cannot be pushed into MATCH (sqlite-vec - // has no join-safe predicate here). Global ANN + post-filter starves small - // eligible sets: they never enter the top-k (#791, #803). Multiplier - // over-fetch alone is not enough either — sqlite-vec caps k at 4096. For a - // filter we therefore exact-scan the eligible vectors when the set is small - // enough, and only then fall back to capped ANN + post-filter. - let vecResults: { hash_seq: string; distance: number }[]; - - if (collectionFilter || filter) { - let eligibleSql = ` - SELECT DISTINCT cv.hash || '_' || cv.seq AS hash_seq - FROM content_vectors cv - JOIN documents d ON d.hash = cv.hash AND d.active = 1 - `; - const eligibleConditions: string[] = []; - const eligibleParams: (string | number)[] = []; - - if (collectionFilter) { - eligibleConditions.push(`d.collection = ?`); - eligibleParams.push(collectionFilter); - } - - if (filter) { - const compiledFilter = compileMetadataFilter(filter, "d"); - eligibleSql += ` JOIN document_metadata dm ON dm.document_id = d.id`; - eligibleConditions.push(`dm.extraction_version = ${METADATA_EXTRACTION_VERSION}`); - eligibleConditions.push(`dm.extraction_error IS NULL`); - eligibleConditions.push(compiledFilter.sql); - eligibleParams.push(...compiledFilter.params); - } - - eligibleSql += ` WHERE ${eligibleConditions.join(" AND ")}`; - - const eligibleHashSeqs: string[] = withLazyContentVectorMigration(db, () => { - const stmt = db.prepare(eligibleSql); - // Large-result query (up to 20k): use iterate() to bound heap, early exit if over max - const seqs: string[] = []; - for (const r of stmt.iterate(...eligibleParams) as IterableIterator<{ hash_seq: string }>) { - seqs.push(r.hash_seq); - if (seqs.length > FILTERED_VEC_EXACT_SCAN_MAX) break; - } - return seqs; - }); - - if (eligibleHashSeqs.length === 0) return []; - - if (eligibleHashSeqs.length <= FILTERED_VEC_EXACT_SCAN_MAX) { - vecResults = exactVecScanByHashSeq(db, embedding, eligibleHashSeqs, limit); - } else { - // Large eligible set: ANN with over-fetch, hard-capped at sqlite-vec's max k. - vecResults = annVecScan(db, embedding, Math.max(limit * 30, limit * 3)); - } - } else { - vecResults = annVecScan(db, embedding, limit * 3); - } - + // Step 1: Get vector matches from sqlite-vec (no JOINs allowed): one KNN per + // collection in scope, three candidate chunks per requested result, so + // multi-chunk documents can still yield `limit` unique files. One statement + // per member rather than `collection_id IN (...)`: the IN form yields k rows + // per value only because SQLite runs vec0's filter once per value, which is + // a planner detail rather than a vec0 contract. + const targets: VecScanTarget[] = collectionIds + ? collectionIds.map(collectionId => ({ collectionId, eligibleRowids: eligible?.get(collectionId) })) + : [{ eligibleRowids: eligible && Array.from(eligible.values()).flat() }]; + const scanTargets = eligible ? targets.filter(t => t.eligibleRowids?.length) : targets; + if (scanTargets.length === 0) return []; + const scan = knnVecScanner(db, collectionIds !== undefined, eligible !== undefined); + const queryVec = new Float32Array(embedding); + const vecResults = scanTargets.flatMap(target => scan(queryVec, limit * 3, target)); if (vecResults.length === 0) return []; - // Step 2: Get chunk info and document data - const hashSeqs = vecResults.map(r => r.hash_seq); - const distanceMap = new Map(vecResults.map(r => [r.hash_seq, r.distance])); - - // One binding for candidate IDs leaves room for a valid near-ceiling - // metadata filter under Node's 32,766-variable limit. The same lookup - // serves exact scans and the capped global fallback. - let docSql = ` + // Step 2: Get chunk info and document data by rowid. The rowids are bound + // as one JSON parameter: thirteen partitions at the k cap exceed SQLite's + // limit on separate bound parameters. + const docParams: SQLiteValue[] = [rowidList(vecResults.map(r => r.rowid))]; + let docFilter = ""; + if (filter) { + // Re-apply the filter on the document join: vectors are content-scoped, + // so one hash can belong to both matching and non-matching documents. + const compiled = compileMetadataFilter(filter, "d"); + docFilter = ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiled.sql}`; + docParams.push(...compiled.params); + } + const distanceByRowid = new Map(vecResults.map(r => [r.rowid, r.distance])); + const docRows = withLazyContentVectorMigration(db, () => db.prepare(` SELECT - cv.hash || '_' || cv.seq as hash_seq, + vr.id AS rowid, cv.hash, cv.pos, 'qmd://' || d.collection || '/' || d.path as filepath, @@ -4704,36 +4713,23 @@ export async function searchVec(db: Database, query: string, model: string, limi d.title, ${cappedBodySql("content.doc")} as body, dm.metadata_json - FROM content_vectors cv - JOIN documents d ON d.hash = cv.hash AND d.active = 1 + FROM json_each(?) j + JOIN ${VEC_ROWS_TABLE} vr ON vr.id = j.value + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + JOIN content_vectors cv ON cv.hash = vr.hash AND cv.seq = vr.seq + JOIN documents d ON d.hash = vr.hash AND d.collection = ci.name JOIN content ON content.hash = d.hash LEFT JOIN document_metadata dm ON dm.document_id = d.id - WHERE cv.hash || '_' || cv.seq IN (SELECT value FROM json_each(?)) - `; - const params: (string | number)[] = [JSON.stringify(hashSeqs)]; - - if (collectionFilter) { - docSql += ` AND d.collection = ?`; - params.push(collectionFilter); - } - - if (filter) { - // Re-apply the filter on the document join: vectors are content-scoped, - // so one hash can belong to both matching and non-matching documents. - const compiledFilter = compileMetadataFilter(filter, "d"); - docSql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`; - params.push(...compiledFilter.params); - } - - const docRows = withLazyContentVectorMigration(db, () => db.prepare(docSql).all(...params) as { - hash_seq: string; hash: string; pos: number; filepath: string; + WHERE d.active = 1${docFilter} + `).all(...docParams) as { + rowid: number; hash: string; pos: number; filepath: string; display_path: string; title: string; body: string; metadata_json: string | null; }[]); // Combine with distances and dedupe by filepath const seen = new Map(); for (const row of docRows) { - const distance = distanceMap.get(row.hash_seq) ?? 1; + const distance = distanceByRowid.get(row.rowid) ?? 1; const existing = seen.get(row.filepath); if (!existing || distance < existing.bestDist) { seen.set(row.filepath, { row, bestDist: distance }); @@ -4810,23 +4806,24 @@ export function getHashesForEmbedding(db: Database, model: string = DEFAULT_EMBE /** * Clear embeddings for the whole index, or just for one collection. * - * When `collection` is omitted the entire content_vectors table is emptied and - * the vectors_vec virtual table is dropped (it is recreated with the right + * When `collection` is omitted the content_vectors and vector_rows tables are + * emptied and the vec0 table is dropped (it is recreated with the right * dimensions on the next embed run). * - * When `collection` is provided, only vectors whose hash is referenced - * exclusively by active documents in that collection are removed. Hashes - * shared with active documents in other collections are left in place so - * vector search keeps working there (content_vectors is keyed globally by - * content hash; identical document bodies across collections share a row). - * vectors_vec is preserved so other collections keep working unless the scoped - * clear empties content_vectors entirely, in which case it is dropped so the - * next embed can recreate the table with the current dimensions. + * When `collection` is provided, only chunks whose hash is referenced + * exclusively by active documents in that collection are removed, from that + * collection's partition and from content_vectors. Hashes shared with active + * documents in other collections are left in place so vector search keeps + * working there (content_vectors is keyed globally by content hash; identical + * document bodies across collections share a row). The vec0 table is dropped + * only when the scoped clear empties content_vectors entirely, so the next + * embed can recreate it with the current dimensions. */ export function clearAllEmbeddings(db: Database, collection?: string): void { if (!collection) { db.exec(`DELETE FROM content_vectors`); - db.exec(`DROP TABLE IF EXISTS vectors_vec`); + db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); + db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); return; } @@ -4841,23 +4838,15 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { AND d2.collection != d.collection ) `; - - const vecTableExists = db - .prepare(`SELECT 1 FROM sqlite_master WHERE type='table' AND name='vectors_vec'`) - .get(); + const collectionId = resolveCollectionId(db, collection); withLazyContentVectorMigration(db, () => { - if (vecTableExists) { - const hashSeqRows = db.prepare(` - SELECT cv.hash, cv.seq - FROM content_vectors cv - WHERE cv.hash IN (${exclusiveHashesQuery}) - `).all(collection) as { hash: string; seq: number }[]; - - const delVec = db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`); - for (const row of hashSeqRows) { - delVec.run(`${row.hash}_${row.seq}`); - } + if (collectionId !== undefined) { + const rows = db.prepare(` + SELECT id FROM ${VEC_ROWS_TABLE} + WHERE collection_id = ? AND hash IN (${exclusiveHashesQuery}) + `).all(collectionId, collection) as { id: number }[]; + deletePartitionRows(db, rows.map((row) => row.id)); } db.prepare(` @@ -4869,20 +4858,17 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { .prepare(`SELECT COUNT(*) AS n FROM content_vectors`) .get() as { n: number }; if (remaining.n === 0) { - db.exec(`DROP TABLE IF EXISTS vectors_vec`); + db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); + db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); } }); } /** - * Insert a single embedding into both content_vectors and vectors_vec tables. - * The hash_seq key is formatted as "hash_seq" for the vectors_vec table. - * - * content_vectors is inserted first so that getHashesForEmbedding (which checks - * only content_vectors) won't re-select the hash on a crash between the two inserts. - * - * vectors_vec uses DELETE + INSERT instead of INSERT OR REPLACE because sqlite-vec's - * vec0 virtual tables silently ignore the OR REPLACE conflict clause. + * Insert a single embedding: the content_vectors row plus one row in the + * partition of every active collection that holds the hash, in one + * transaction so a crash cannot leave the tables out of step. A hash with no + * active document gets its content_vectors row only. */ export function insertEmbedding( db: Database, @@ -4895,18 +4881,15 @@ export function insertEmbedding( totalChunks: number = 1, fingerprint: string = getEmbeddingFingerprint(model) ): void { - const hashSeq = `${hash}_${seq}`; - withLazyContentVectorMigration(db, () => { - // Insert content_vectors first — crash-safe ordering (see getHashesForEmbedding) - const insertContentVectorStmt = db.prepare(`INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at) VALUES (?, ?, ?, ?, ?, ?, ?)`); - insertContentVectorStmt.run(hash, seq, pos, model, fingerprint, totalChunks, embeddedAt); - - // vec0 virtual tables don't support OR REPLACE — use DELETE + INSERT - const deleteVecStmt = db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`); - const insertVecStmt = db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`); - deleteVecStmt.run(hashSeq); - insertVecStmt.run(hashSeq, embedding); + db.transaction(() => { + db.prepare(`INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at) VALUES (?, ?, ?, ?, ?, ?, ?)`) + .run(hash, seq, pos, model, fingerprint, totalChunks, embeddedAt); + const collections = db.prepare(`SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`).all(hash) as { collection: string }[]; + for (const { collection } of collections) { + upsertPartitionVector(db, hash, seq, allocateCollectionId(db, collection), embedding); + } + })(); }); } @@ -4915,15 +4898,13 @@ function removeIncompleteEmbeddings(db: Database, expectedChunksByHash: Map row.id)); deleteContentStmt.run(hash, model); removed += rows.length; } @@ -5688,7 +5669,7 @@ export function getStatusSummary(db: Database, model: string = DEFAULT_EMBED_MOD const totalDocs = (db.prepare(`SELECT COUNT(*) as c FROM documents WHERE active = 1`).get() as { c: number }).c; const needsEmbedding = getHashesNeedingEmbedding(db, undefined, model); - const hasVectors = !!db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get(); + const hasVectors = hasVectorIndex(db); return { totalDocuments: totalDocs, @@ -5969,9 +5950,7 @@ export async function hybridQuery( const rankedLists: RankedResult[][] = []; const rankedListMeta: RankedListMeta[] = []; const docidMap = new Map(); // filepath -> docid - const hasVectors = !!store.db.prepare( - `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` - ).get(); + const hasVectors = hasVectorIndex(store.db); // Step 1: BM25 probe — strong signal skips expensive LLM expansion // When intent is provided, disable strong-signal bypass — the obvious BM25 @@ -6292,9 +6271,7 @@ export async function vectorSearchQuery( const filter = options?.filter; const intent = options?.intent; - const hasVectors = !!store.db.prepare( - `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` - ).get(); + const hasVectors = hasVectorIndex(store.db); if (!hasVectors) return []; // Expand query — filter to vec/hyde only (lex queries target FTS, not vector) @@ -6413,9 +6390,7 @@ export async function structuredSearch( const rankedLists: RankedResult[][] = []; const rankedListMeta: RankedListMeta[] = []; const docidMap = new Map(); // filepath -> docid - const hasVectors = !!store.db.prepare( - `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` - ).get(); + const hasVectors = hasVectorIndex(store.db); // Each search yields ONE ranked list over the union of the named collections // (undefined = all). searchFTS/searchVec merge per-collection results by score, diff --git a/src/vec-layout.ts b/src/vec-layout.ts index f61c0c310..ba3328c81 100644 --- a/src/vec-layout.ts +++ b/src/vec-layout.ts @@ -175,6 +175,87 @@ export function rowidList(rowids: readonly number[]): string { return JSON.stringify(rowids); } +/** Deletes vector rows by rowid from the vec0 table (when it exists) and the mapping. */ +export function deletePartitionRows(db: Database, rowids: readonly number[]): void { + if (rowids.length === 0) return; + const deleteVec = hasVectorIndex(db) ? db.prepare(`DELETE FROM ${VEC_TABLE} WHERE rowid = ?`) : null; + const deleteRow = db.prepare(`DELETE FROM ${VEC_ROWS_TABLE} WHERE id = ?`); + for (const id of rowids) { + deleteVec?.run(vecInteger(id)); + deleteRow.run(id); + } +} + +/** + * Replace or insert the vector of (hash, seq) in one collection's partition. + * vec0 ignores OR REPLACE, so an existing row is deleted and its rowid reused. + */ +export function upsertPartitionVector(db: Database, hash: string, seq: number, collectionId: number, embedding: Float32Array | Uint8Array): void { + const existing = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? AND collection_id = ?`) + .get(hash, seq, collectionId) as { id: number } | undefined; + let rowid: number; + if (existing) { + db.prepare(`DELETE FROM ${VEC_TABLE} WHERE rowid = ?`).run(vecInteger(existing.id)); + rowid = existing.id; + } else { + rowid = Number(db.prepare(`INSERT INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES (?, ?, ?)`).run(hash, seq, collectionId).lastInsertRowid); + } + db.prepare(`INSERT INTO ${VEC_TABLE} (rowid, collection_id, embedding) VALUES (?, ?, ?)`).run(vecInteger(rowid), vecInteger(collectionId), embedding); +} + +/** Bulk writer that inserts a vector under a collection unless that partition row exists. */ +export class PartitionWriter { + private readonly ids = new Map(); + private readonly collectionsOf; + private readonly insertRow; + private readonly insertVec; + + constructor(private readonly db: Database) { + this.collectionsOf = db.prepare(`SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`); + this.insertRow = db.prepare(`INSERT OR IGNORE INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES (?, ?, ?)`); + this.insertVec = db.prepare(`INSERT INTO ${VEC_TABLE} (rowid, collection_id, embedding) VALUES (?, ?, ?)`); + } + + private idOf(collection: string): number { + let id = this.ids.get(collection); + if (id === undefined) { + id = allocateCollectionId(this.db, collection); + this.ids.set(collection, id); + } + return id; + } + + /** True when a row was written; false when the partition already held (hash, seq). */ + writeToCollection(hash: string, seq: number, collection: string, embedding: Uint8Array | Float32Array): boolean { + const id = this.idOf(collection); + const inserted = this.insertRow.run(hash, seq, id); + if (inserted.changes === 0) return false; + this.insertVec.run(vecInteger(inserted.lastInsertRowid), vecInteger(id), embedding); + return true; + } + + writeToActiveCollections(hash: string, seq: number, embedding: Uint8Array | Float32Array): number { + let written = 0; + for (const row of this.collectionsOf.all(hash) as { collection: string }[]) { + if (this.writeToCollection(hash, seq, row.collection, embedding)) written++; + } + return written; + } +} + +/** Stored vector bytes of (hash, seq) from whichever partition holds it. */ +export function storedEmbedding(db: Database, hash: string, seq: number): Uint8Array | undefined { + const row = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? LIMIT 1`).get(hash, seq) as { id: number } | undefined; + if (!row) return undefined; + const stored = db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`).get(vecInteger(row.id)) as { embedding: Uint8Array } | undefined; + return stored?.embedding; +} + +/** (hash, seq) behind a vec0 rowid. */ +export function partitionRowKey(db: Database, rowid: number): { hash: string; seq: number } | undefined { + return db.prepare(`SELECT hash, seq FROM ${VEC_ROWS_TABLE} WHERE id = ?`).get(rowid) as { hash: string; seq: number } | undefined; +} + export type MissingPartitionRow = { hash: string; seq: number; collection: string }; /** diff --git a/test/eval.test.ts b/test/eval.test.ts index d575ff847..1e2dfa4d8 100644 --- a/test/eval.test.ts +++ b/test/eval.test.ts @@ -15,6 +15,7 @@ import { mkdtempSync, rmSync, readFileSync, readdirSync } from "fs"; import { join } from "path"; import { tmpdir } from "os"; import { openDatabase } from "../src/db.js"; +import { VEC_TABLE, hasVectorIndex } from "../src/vec-layout.js"; import type { Database } from "../src/db.js"; import { createHash } from "crypto"; import { fileURLToPath } from "url"; @@ -166,12 +167,8 @@ describe.skipIf(!!process.env.CI)("Vector Search", () => { db = store.db; // Check if embeddings already exist (from previous test run) - const vecTable = db.prepare( - `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` - ).get(); - - if (vecTable) { - const count = db.prepare(`SELECT COUNT(*) as cnt FROM vectors_vec`).get() as { cnt: number }; + if (hasVectorIndex(db)) { + const count = db.prepare(`SELECT COUNT(*) as cnt FROM ${VEC_TABLE}`).get() as { cnt: number }; if (count.cnt > 0) { hasEmbeddings = true; return; @@ -277,11 +274,8 @@ describe.skipIf(!!process.env.CI)("Hybrid Search (RRF)", () => { store = createStore(); db = store.db; // Check if vectors exist - const vecTable = db.prepare( - `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'` - ).get(); - if (vecTable) { - const count = db.prepare(`SELECT COUNT(*) as cnt FROM vectors_vec`).get() as { cnt: number }; + if (hasVectorIndex(db)) { + const count = db.prepare(`SELECT COUNT(*) as cnt FROM ${VEC_TABLE}`).get() as { cnt: number }; hasVectors = count.cnt > 0; } }); diff --git a/test/mcp.test.ts b/test/mcp.test.ts index 2a17ba84e..4a3b9bac5 100644 --- a/test/mcp.test.ts +++ b/test/mcp.test.ts @@ -18,6 +18,8 @@ import type { CollectionConfig } from "../src/collections"; import { setConfigIndexName } from "../src/collections"; import { syncConfigToDb } from "../src/store"; import { initializeMetadataSchema } from "../src/metadata-store"; +import { VEC_TABLE, createVectorMetadataTables, vecLayout } from "../src/vec-layout"; +import { migrateVectorLayout } from "../src/store-migrations"; // ============================================================================= // Test Database Setup @@ -128,8 +130,10 @@ function initTestDatabase(db: Database): void { // Document metadata tables — searchFTS/searchVec join them for result metadata initializeMetadataSchema(db); + createVectorMetadataTables(db); } +/** Seeds the legacy vector table, then runs the layout migration on it. */ function seedTestData(db: Database): void { const now = new Date().toISOString(); @@ -192,6 +196,8 @@ function seedTestData(db: Database): void { db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embed_fingerprint, embedded_at) VALUES (?, 0, 0, ?, ?, ?)`).run(doc.hash, DEFAULT_EMBED_MODEL, getEmbeddingFingerprint(DEFAULT_EMBED_MODEL), now); db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${doc.hash}_0`, embedding); } + expect(migrateVectorLayout(db, { sqliteVecAvailable: true })).toBe("applied"); + expect(vecLayout(db)).toMatchObject({ kind: "partitioned", dimensions: 768 }); } // ============================================================================= @@ -342,6 +348,7 @@ describe("MCP Server", () => { const emptyDb = openDatabase(":memory:"); initTestDatabase(emptyDb); emptyDb.exec("DROP TABLE IF EXISTS vectors_vec"); + emptyDb.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); const results = await searchVec(emptyDb, "test", DEFAULT_EMBED_MODEL, 10); expect(results.length).toBe(0); diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index a819cbcb6..9d973d086 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -283,6 +283,49 @@ describe("searchVec with metadata filter", () => { ); expect(filtered.map(r => r.displayPath)).toEqual(["docs/b.md"]); }); + + test("a filter admitting more than 20,000 chunks still returns its nearest eligible documents", async () => { + store.ensureVecTable(3); + store.db.exec("BEGIN"); + for (let i = 0; i < 20_001; i++) { + await insertEmbeddedDoc("book", `eligible-${i}.md`, `# Eligible ${i}`, [0, 1, 0], { eligible: true }); + } + for (let i = 0; i < 200; i++) { + await insertEmbeddedDoc("book", `closer-${i}.md`, `# Closer ${i}`, [1, 0, 0], { eligible: false }); + } + store.db.exec("COMMIT"); + + const filtered = await searchVec( + store.db, "q", model, 5, "book", undefined, queryEmbedding, undefined, + { key: "eligible", operator: "eq", value: true }, + ); + expect(filtered).toHaveLength(5); + expect(filtered.every(r => r.metadata.eligible === true)).toBe(true); + }, 120_000); + + test("shared content hash within one collection returns only the matching document path", async () => { + store.ensureVecTable(3); + const now = new Date().toISOString(); + const body = "# Shared body"; + const hash = await hashContent(body); + insertContent(store.db, hash, body, now); + + const publishedId = insertDocument(store.db, "notes", "published-copy.md", "t", hash, now, now); + const draftId = insertDocument(store.db, "notes", "draft-copy.md", "t", hash, now, now); + replaceDocumentMetadata(store.db, publishedId, { + metadata: { status: "published" }, extractionVersion: METADATA_EXTRACTION_VERSION, + }); + replaceDocumentMetadata(store.db, draftId, { + metadata: { status: "draft" }, extractionVersion: METADATA_EXTRACTION_VERSION, + }); + insertEmbedding(store.db, hash, 0, 0, new Float32Array([1, 0, 0]), model, now, 1); + + const filtered = await searchVec( + store.db, "q", model, 10, "notes", undefined, queryEmbedding, undefined, + { key: "status", operator: "eq", value: "published" }, + ); + expect(filtered.map(r => r.displayPath)).toEqual(["notes/published-copy.md"]); + }); }); describe("structuredSearch with metadata filter", () => { diff --git a/test/multi-collection-filter.test.ts b/test/multi-collection-filter.test.ts index 63aae7d5b..ac779178e 100644 --- a/test/multi-collection-filter.test.ts +++ b/test/multi-collection-filter.test.ts @@ -1,9 +1,10 @@ /** * Unit tests for multi-collection CLI filter input (#191, #775). * - * Collection scoping is applied in searchFTS/searchVec (search each collection, - * then merge). These tests cover CLI parseArgs / input normalization only. - * Starvation regressions live in store.test.ts. + * Collection scoping is applied in searchFTS (search each collection, then + * merge) and searchVec (one KNN per collection partition, merged by distance). + * These tests cover CLI parseArgs / input normalization only. Starvation + * regressions live in store.test.ts. */ import { describe, test, expect } from "vitest"; diff --git a/test/store.helpers.unit.test.ts b/test/store.helpers.unit.test.ts index aaaf3a0ee..0a058e2cd 100644 --- a/test/store.helpers.unit.test.ts +++ b/test/store.helpers.unit.test.ts @@ -6,6 +6,7 @@ import { describe, test, expect } from "vitest"; import { mkdtempSync, mkdirSync, writeFileSync, symlinkSync, rmSync, chmodSync } from "node:fs"; import { join } from "node:path"; import { tmpdir } from "node:os"; +import { VEC_TABLE } from "../src/vec-layout.js"; import { homedir, resolve, @@ -169,10 +170,11 @@ describe("countOrphanedVectors", () => { describe("cleanupOrphanedVectors", () => { test("returns 0 when vec table exists in schema but sqlite-vec is unavailable", () => { const prepare = (sql: string) => { - if (sql.includes("sqlite_master") && sql.includes("vectors_vec")) { - return { get: () => ({ name: "vectors_vec" }) }; + if (sql.includes("sqlite_master")) { + const row = { name: VEC_TABLE, sql: `CREATE VIRTUAL TABLE ${VEC_TABLE} USING vec0(collection_id INTEGER PARTITION KEY, embedding float[3] distance_metric=cosine)` }; + return { get: () => row, all: () => [row] }; } - if (sql.includes("SELECT 1 FROM vectors_vec LIMIT 0")) { + if (sql.includes(`SELECT 1 FROM ${VEC_TABLE} LIMIT 0`)) { return { get: () => { throw new Error("no such module: vec0"); } }; } throw new Error(`Unexpected SQL in test: ${sql}`); diff --git a/test/store.test.ts b/test/store.test.ts index 6f9b63578..0a0f92856 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -52,14 +52,15 @@ import { isDocid, syncConfigToDb, reindexCollection, - removeCollection, - renameCollection, resolveVirtualPath, STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP, insertContent, insertDocument, cleanupOrphanedVectors, + clearAllEmbeddings, + removeCollection, + renameCollection, generateEmbeddings, maybeAdoptLegacyEmbeddingFingerprint, getHybridRrfWeights, @@ -74,6 +75,7 @@ import { type RankedListMeta, } from "../src/store.js"; import type { CollectionConfig } from "../src/collections.js"; +import { VEC_ROWS_TABLE, VEC_TABLE, deletePartitionRows, resolveCollectionId } from "../src/vec-layout.js"; // ============================================================================= // LlamaCpp Setup @@ -3850,28 +3852,30 @@ describe("Index Status", () => { }); describe("cleanupOrphanedVectors atomicity", () => { - // Seeds one active document (1 chunk) and one inactive document (2 chunks), - // so cleanup should remove exactly the 2 orphaned chunks from both tables. + // Seeds one active document (1 chunk) and one document (2 chunks) that goes + // inactive after embedding, so cleanup should remove exactly the 2 orphaned + // chunks from both tables. async function seedOrphanFixture(store: Store): Promise { const collectionName = await createTestCollection(); const now = new Date().toISOString(); store.ensureVecTable(3); await insertTestDocument(store.db, collectionName, { name: "kept-doc", hash: "keephash" }); - await insertTestDocument(store.db, collectionName, { name: "orphaned-doc", hash: "orphanhash", active: 0 }); + await insertTestDocument(store.db, collectionName, { name: "orphaned-doc", hash: "orphanhash" }); store.insertEmbedding("keephash", 0, 0, new Float32Array([1, 2, 3]), "test-model", now, 1); store.insertEmbedding("orphanhash", 0, 0, new Float32Array([4, 5, 6]), "test-model", now, 2); store.insertEmbedding("orphanhash", 1, 10, new Float32Array([7, 8, 9]), "test-model", now, 2); + store.db.prepare(`UPDATE documents SET active = 0 WHERE hash = 'orphanhash'`).run(); } function vecCounts(db: Database): { vec: number; meta: number } { - const vec = (db.prepare(`SELECT COUNT(*) AS c FROM vectors_vec`).get() as { c: number }).c; + const vec = (db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get() as { c: number }).c; const meta = (db.prepare(`SELECT COUNT(*) AS c FROM content_vectors`).get() as { c: number }).c; return { vec, meta }; } // Fault injection: same connection, but the content_vectors DELETE throws — - // after the vectors_vec DELETE already executed inside the transaction. + // after the vector DELETEs already executed inside the transaction. function makeFailingDb(db: Database): Database { return { prepare: (sql: string) => db.prepare(sql), @@ -3898,25 +3902,25 @@ describe("cleanupOrphanedVectors atomicity", () => { expect(vecCounts(store.db)).toEqual({ vec: 1, meta: 1 }); const survivor = store.db.prepare(`SELECT hash FROM content_vectors`).get() as { hash: string }; expect(survivor.hash).toBe("keephash"); - const survivorVec = store.db.prepare(`SELECT hash_seq FROM vectors_vec`).get() as { hash_seq: string }; - expect(survivorVec.hash_seq).toBe("keephash_0"); + const survivorVec = store.db.prepare(`SELECT hash, seq FROM ${VEC_ROWS_TABLE}`).get() as { hash: string; seq: number }; + expect(survivorVec).toEqual({ hash: "keephash", seq: 0 }); } finally { await cleanupTestDb(store); } }); - test("rolls back the vectors_vec DELETE when the content_vectors DELETE fails", async () => { + test("rolls back the vector DELETEs when the content_vectors DELETE fails", async () => { const store = await createTestStore(); try { await seedOrphanFixture(store); const db = store.db; - // Without the transaction wrap this used to leave vectors_vec already + // Without the transaction wrap this used to leave the vector table already // purged while content_vectors still claimed the chunks were embedded // (silent desync). expect(() => cleanupOrphanedVectors(makeFailingDb(db))).toThrow("injected failure between deletes"); - // Both tables must be untouched — the vectors_vec DELETE was rolled back. + // Both tables must be untouched — the vector DELETEs were rolled back. expect(vecCounts(db)).toEqual({ vec: 3, meta: 3 }); // The connection is left in a clean state: a plain retry succeeds. @@ -3966,7 +3970,7 @@ describe("cleanupOrphanedVectors atomicity", () => { } } // If the cleanup ran inline instead of inside its own savepoint, the - // vectors_vec DELETE would survive the caught failure and commit with + // vector DELETEs would survive the caught failure and commit with // the outer transaction below. db.prepare(`INSERT INTO content (hash, doc, created_at) VALUES (?, ?, ?)`) .run("outer-survivor", "outer doc", new Date().toISOString()); @@ -4104,7 +4108,7 @@ describe("Vector Table", () => { // Initially no vector table let exists = store.db.prepare(` - SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec' + SELECT name FROM sqlite_master WHERE type='table' AND name='${VEC_TABLE}' `).get(); expect(exists).toBeFalsy(); // null or undefined @@ -4112,7 +4116,7 @@ describe("Vector Table", () => { store.ensureVecTable(768); exists = store.db.prepare(` - SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec' + SELECT name FROM sqlite_master WHERE type='table' AND name='${VEC_TABLE}' `).get(); expect(exists).toBeTruthy(); @@ -4127,7 +4131,7 @@ describe("Vector Table", () => { // Check dimensions const tableInfo = store.db.prepare(` - SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec' + SELECT sql FROM sqlite_master WHERE type='table' AND name='${VEC_TABLE}' `).get() as { sql: string }; expect(tableInfo.sql).toContain("float[768]"); @@ -4136,36 +4140,37 @@ describe("Vector Table", () => { // Original table should still exist untouched const tableInfoAfter = store.db.prepare(` - SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec' + SELECT sql FROM sqlite_master WHERE type='table' AND name='${VEC_TABLE}' `).get() as { sql: string }; expect(tableInfoAfter.sql).toContain("float[768]"); await cleanupTestDb(store); }); - test("insertEmbedding is idempotent for an existing vec0 hash_seq (#598)", async () => { + test("insertEmbedding replaces the stored vector of an existing chunk (#598)", async () => { const store = await createTestStore(); + const collection = await createTestCollection(); store.ensureVecTable(2); const hash = "existinghashseq"; const first = new Float32Array([0.1, 0.2]); const second = new Float32Array([0.3, 0.4]); const now = new Date().toISOString(); + await insertTestDocument(store.db, collection, { name: "doc", hash }); + store.insertEmbedding(hash, 0, 0, first, "test-model", now); + const rowid = (store.db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = 0`).get(hash) as { id: number }).id; - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_0`, first); - - // Reproduces sqlite-vec's broken conflict handling: vec0 does not honor OR REPLACE. - expect(() => { - store.db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_0`, second); - }).toThrow(/UNIQUE constraint failed/i); - - // QMD must therefore use DELETE + INSERT when upserting the vector row. + // vec0 does not honor OR REPLACE: the partition row is deleted and + // re-inserted under the same rowid. expect(() => store.insertEmbedding(hash, 0, 0, second, "test-model", now)).not.toThrow(); - const vectorCount = store.db.prepare(`SELECT COUNT(*) AS count FROM vectors_vec WHERE hash_seq = ?`).get(`${hash}_0`) as { count: number }; + expect(store.db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = 0`).all(hash)).toEqual([{ id: rowid }]); + const vectorCount = store.db.prepare(`SELECT COUNT(*) AS count FROM ${VEC_TABLE}`).get() as { count: number }; const metadataCount = store.db.prepare(`SELECT COUNT(*) AS count FROM content_vectors WHERE hash = ? AND seq = 0`).get(hash) as { count: number }; expect(vectorCount.count).toBe(1); expect(metadataCount.count).toBe(1); + const stored = store.db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`).get(BigInt(rowid)) as { embedding: Uint8Array }; + expect(Array.from(new Float32Array(stored.embedding.buffer, stored.embedding.byteOffset, 2))).toEqual(Array.from(second)); await cleanupTestDb(store); }); @@ -4327,10 +4332,10 @@ describe("Vector Search collection filter", () => { queryEmbedding[0] = 1; // 250 nearer neighbours in the large collection. With limit=3: - // - old global k=limit*3=9 never sees `small` - // - a plain multiplier (limit*30=90) still misses it - // - sqlite-vec also caps k at 4096, so multipliers cannot fix tiny - // collections in huge indexes. Collection-scoped exact scan does. + // - an unscoped top-k (k=limit*3=9) never sees `small` + // - sqlite-vec caps k at 4096, so a wider over-fetch cannot fix tiny + // collections in huge indexes; a KNN inside the collection's + // partition does. for (let i = 0; i < 250; i++) { const hash = `largehash${String(i).padStart(3, "0")}`; await insertTestDocument(store.db, large, { @@ -4341,8 +4346,7 @@ describe("Vector Search collection filter", () => { }); const embedding = new Float32Array(dims); embedding[0] = 1; - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(hash, now); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_0`, embedding); + store.insertEmbedding(hash, 0, 0, embedding, 'test', new Date().toISOString()); } const targetHash = "smallhash001"; @@ -4352,12 +4356,11 @@ describe("Vector Search collection filter", () => { body: "Target document in the small collection", displayPath: "target.md", }); - // Farther than the noise vectors — only found via collection-scoped scan. + // Farther than the noise vectors; only a scan of the small partition reaches it. const targetEmbedding = new Float32Array(dims); targetEmbedding[0] = 0.6; targetEmbedding[1] = 0.8; - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(targetHash, now); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${targetHash}_0`, targetEmbedding); + store.insertEmbedding(targetHash, 0, 0, targetEmbedding, 'test', new Date().toISOString()); const filtered = await store.searchVec( "ignored — embedding precomputed", @@ -4408,8 +4411,7 @@ describe("Vector Search collection filter", () => { }); const embedding = new Float32Array(dims); embedding[0] = 1; - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(hash, now); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_0`, embedding); + store.insertEmbedding(hash, 0, 0, embedding, 'test', new Date().toISOString()); } const kbHash = "kbhash001"; @@ -4422,8 +4424,7 @@ describe("Vector Search collection filter", () => { const kbEmbedding = new Float32Array(dims); kbEmbedding[0] = 0.55; kbEmbedding[1] = 0.84; - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(kbHash, now); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${kbHash}_0`, kbEmbedding); + store.insertEmbedding(kbHash, 0, 0, kbEmbedding, 'test', new Date().toISOString()); const notesHash = "noteshash001"; await insertTestDocument(store.db, notes, { @@ -4435,8 +4436,7 @@ describe("Vector Search collection filter", () => { const notesEmbedding = new Float32Array(dims); notesEmbedding[0] = 0.5; notesEmbedding[1] = 0.87; - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(notesHash, now); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${notesHash}_0`, notesEmbedding); + store.insertEmbedding(notesHash, 0, 0, notesEmbedding, 'test', new Date().toISOString()); const global = await store.searchVec( "ignored — embedding precomputed", @@ -4500,6 +4500,401 @@ describe("Vector Search collection filter", () => { await cleanupTestDb(store); }); + + const DIMS = 8; + const query: number[] = Array(DIMS).fill(0); + query[0] = 1; + + function vector(x: number, y: number): Float32Array { + const embedding = new Float32Array(DIMS); + embedding[0] = x; + embedding[1] = y; + return embedding; + } + + /** Chunk `seq` of `hash` sits at pos seq * 100, with one partition row per active collection of the hash. */ + function insertChunkVectors(store: Store, hash: string, chunks: readonly Float32Array[]): void { + const now = new Date().toISOString(); + chunks.forEach((embedding, seq) => store.insertEmbedding(hash, seq, seq * 100, embedding, "test", now, chunks.length)); + } + + async function insertVecDoc(store: Store, collection: string, hash: string, chunks: readonly Float32Array[]): Promise { + await insertTestDocument(store.db, collection, { name: hash, hash, body: `Document ${hash}`, displayPath: `${hash}.md` }); + insertChunkVectors(store, hash, chunks); + } + + /** Documents orthogonal to the query. */ + async function insertFillerDocs(store: Store, collection: string, prefix: string, count: number): Promise { + for (let i = 0; i < count; i++) { + await insertVecDoc(store, collection, `${prefix}${String(i).padStart(2, "0")}`, [vector(0, 1)]); + } + } + + function vectorRowCount(store: Store, collection: string): number { + const id = resolveCollectionId(store.db, collection); + if (id === undefined) return 0; + return (store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE} WHERE collection_id = ?`).get(id) as { c: number }).c; + } + + test("searchVec scoped to two of three collections returns exactly those two", async () => { + const store = await createTestStore(); + const first = await createTestCollection({ name: "first", pwd: "/test/first" }); + const second = await createTestCollection({ name: "second", pwd: "/test/second" }); + const third = await createTestCollection({ name: "third", pwd: "/test/third" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, first, "firsthash", [vector(1, 0)]); + await insertVecDoc(store, second, "secondhash", [vector(0.9, 0.44)]); + await insertVecDoc(store, third, "thirdhash", [vector(0.8, 0.6)]); + await insertFillerDocs(store, second, "secondfill", 4); + + const results = await store.searchVec("ignored", "test-model", 3, [first, third], undefined, query); + expect(results.map((r) => r.collectionName).sort()).toEqual([first, third].sort()); + + await cleanupTestDb(store); + }); + + test("searchVec runs one KNN statement per collection in scope and an unpartitioned one otherwise", async () => { + const store = await createTestStore(); + const first = await createTestCollection({ name: "first", pwd: "/test/first" }); + const second = await createTestCollection({ name: "second", pwd: "/test/second" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, first, "firsthash", [vector(1, 0)]); + await insertVecDoc(store, second, "secondhash", [vector(0.6, 0.8)]); + const prepare = vi.spyOn(store.db, "prepare"); + const knnStatements = () => prepare.mock.calls.map((call) => String(call[0])).filter((sql) => sql.includes("MATCH")); + try { + const scoped = await store.searchVec("ignored", "test-model", 3, [first, second], undefined, query); + expect(scoped.map((r) => r.hash)).toEqual(["firsthash", "secondhash"]); + expect(knnStatements()).toHaveLength(1); + expect(knnStatements()[0]).toContain("collection_id = ?"); + expect(knnStatements()[0]).not.toContain(" IN "); + + prepare.mockClear(); + await store.searchVec("ignored", "test-model", 3, undefined, undefined, query); + expect(knnStatements()).toHaveLength(1); + expect(knnStatements()[0]).not.toContain("collection_id"); + } finally { + prepare.mockRestore(); + } + + await cleanupTestDb(store); + }); + + test("searchVec returns a hash shared with an out-of-scope collection once, named for the scoped collection", async () => { + const store = await createTestStore(); + const included = await createTestCollection({ name: "included", pwd: "/test/included" }); + const excluded = await createTestCollection({ name: "excluded", pwd: "/test/excluded" }); + store.ensureVecTable(DIMS); + await insertTestDocument(store.db, excluded, { name: "sharedhash", hash: "sharedhash", body: "Document sharedhash", displayPath: "sharedhash.md" }); + await insertTestDocument(store.db, included, { name: "copy", hash: "sharedhash", body: "Document sharedhash", displayPath: "copy.md" }); + insertChunkVectors(store, "sharedhash", [vector(1, 0)]); + await insertFillerDocs(store, excluded, "excludedfill", 3); + + const results = await store.searchVec("ignored", "test-model", 5, included, undefined, query); + expect(results).toHaveLength(1); + expect(results[0]!.collectionName).toBe(included); + expect(results[0]!.displayPath).toBe(`${included}/copy.md`); + + await cleanupTestDb(store); + }); + + test("searchVec returns a hash shared by two in-scope collections once per collection", async () => { + const store = await createTestStore(); + const alpha = await createTestCollection({ name: "alpha", pwd: "/test/alpha" }); + const beta = await createTestCollection({ name: "beta", pwd: "/test/beta" }); + store.ensureVecTable(DIMS); + await insertTestDocument(store.db, alpha, { name: "shared", hash: "sharedhash", body: "Document sharedhash", displayPath: "shared.md" }); + await insertTestDocument(store.db, beta, { name: "shared", hash: "sharedhash", body: "Document sharedhash", displayPath: "shared.md" }); + insertChunkVectors(store, "sharedhash", [vector(1, 0)]); + + const results = await store.searchVec("ignored", "test-model", 5, [alpha, beta], undefined, query); + expect(results.map((r) => r.collectionName).sort()).toEqual([alpha, beta].sort()); + expect(new Set(results.map((r) => r.filepath)).size).toBe(2); + + await cleanupTestDb(store); + }); + + test("searchVec ignores a hash whose only in-scope row is inactive", async () => { + const store = await createTestStore(); + const included = await createTestCollection({ name: "included", pwd: "/test/included" }); + const excluded = await createTestCollection({ name: "excluded", pwd: "/test/excluded" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, excluded, "sharedhash", [vector(1, 0)]); + await insertTestDocument(store.db, included, { name: "stale", hash: "sharedhash", body: "Document sharedhash", displayPath: "stale.md", active: 0 }); + + const results = await store.searchVec("ignored", "test-model", 5, included, undefined, query); + expect(results).toEqual([]); + + await cleanupTestDb(store); + }); + + test("searchVec returns a document once when its hash has an active and an inactive row in scope", async () => { + const store = await createTestStore(); + const included = await createTestCollection({ name: "included", pwd: "/test/included" }); + const outside = await createTestCollection({ name: "outside", pwd: "/test/outside" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, included, "duphash", [vector(1, 0)]); + await insertFillerDocs(store, outside, "outsidefill", 3); + await insertTestDocument(store.db, included, { name: "stale", hash: "duphash", body: "Document duphash", displayPath: "stale.md", active: 0 }); + + const results = await store.searchVec("ignored", "test-model", 5, included, undefined, query); + expect(results).toHaveLength(1); + expect(results[0]!.displayPath).toBe(`${included}/duphash.md`); + + await cleanupTestDb(store); + }); + + test("searchVec returns empty for a scoped collection that has documents but no vectors", async () => { + const store = await createTestStore(); + const embedded = await createTestCollection({ name: "embedded", pwd: "/test/embedded" }); + const bare = await createTestCollection({ name: "bare", pwd: "/test/bare" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, embedded, "embeddedhash", [vector(1, 0)]); + await insertFillerDocs(store, embedded, "embeddedfill", 2); + await insertTestDocument(store.db, bare, { name: "plain", body: "Not embedded", displayPath: "plain.md" }); + + const results = await store.searchVec("ignored", "test-model", 3, bare, undefined, query); + expect(results).toEqual([]); + + await cleanupTestDb(store); + }); + + test("searchVec returns empty for a collection name that no document carries", async () => { + const store = await createTestStore(); + const embedded = await createTestCollection({ name: "embedded", pwd: "/test/embedded" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, embedded, "embeddedhash", [vector(1, 0)]); + + expect(await store.searchVec("ignored", "test-model", 3, "never-indexed", undefined, query)).toEqual([]); + const mixed = await store.searchVec("ignored", "test-model", 3, ["never-indexed", embedded], undefined, query); + expect(mixed.map((r) => r.hash)).toEqual(["embeddedhash"]); + + await cleanupTestDb(store); + }); + + test("searchVec accepts a scoped limit above sqlite-vec's k cap", async () => { + const store = await createTestStore(); + const embedded = await createTestCollection({ name: "embedded", pwd: "/test/embedded" }); + const outside = await createTestCollection({ name: "outside", pwd: "/test/outside" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, embedded, "nearhash", [vector(1, 0)]); + await insertVecDoc(store, embedded, "farhash", [vector(0.6, 0.8)]); + await insertFillerDocs(store, outside, "outsidefill", 4); + + const scoped = await store.searchVec("ignored", "test-model", 2000, embedded, undefined, query); + expect(scoped.map((r) => r.hash)).toEqual(["nearhash", "farhash"]); + const unscoped = await store.searchVec("ignored", "test-model", 2000, undefined, undefined, query); + expect(unscoped).toHaveLength(6); + + await cleanupTestDb(store); + }); + + test("searchVec treats an empty collection list like no scope", async () => { + const store = await createTestStore(); + const first = await createTestCollection({ name: "first", pwd: "/test/first" }); + const second = await createTestCollection({ name: "second", pwd: "/test/second" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, first, "firsthash", [vector(1, 0)]); + await insertVecDoc(store, second, "secondhash", [vector(0.6, 0.8)]); + + const unscoped = await store.searchVec("ignored", "test-model", 3, undefined, undefined, query); + const emptyScope = await store.searchVec("ignored", "test-model", 3, [], undefined, query); + expect(unscoped.map((r) => r.hash)).toEqual(["firsthash", "secondhash"]); + expect(emptyScope).toEqual(unscoped); + + await cleanupTestDb(store); + }); + + test("searchVec collapses a scoped over-fetch to one row per file at its best chunk", async () => { + const store = await createTestStore(); + const alpha = await createTestCollection({ name: "alpha", pwd: "/test/alpha" }); + const beta = await createTestCollection({ name: "beta", pwd: "/test/beta" }); + const outside = await createTestCollection({ name: "outside", pwd: "/test/outside" }); + store.ensureVecTable(DIMS); + await insertFillerDocs(store, outside, "outsidefill", 30); + const near = vector(1, 0); + const far = vector(0.6, 0.8); + for (let i = 0; i < 3; i++) { + await insertVecDoc(store, alpha, `long${i}`, Array.from({ length: 10 }, () => near)); + } + for (let i = 0; i < 20; i++) { + await insertVecDoc(store, i % 2 === 0 ? alpha : beta, `short${String(i).padStart(2, "0")}`, [far]); + } + + const results = await store.searchVec("ignored", "test-model", 10, [alpha, beta], undefined, query); + expect(results).toHaveLength(10); + expect(results.slice(0, 3).map((r) => r.hash).sort()).toEqual(["long0", "long1", "long2"]); + expect(new Set(results.map((r) => r.filepath)).size).toBe(results.length); + + await cleanupTestDb(store); + }); + + test("searchVec rows carry the vec source, a cosine score, and the best chunk position", async () => { + const store = await createTestStore(); + const embedded = await createTestCollection({ name: "embedded", pwd: "/test/embedded" }); + const outside = await createTestCollection({ name: "outside", pwd: "/test/outside" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, embedded, "twochunks", [vector(0, 1), vector(0.6, 0.8)]); + await insertFillerDocs(store, outside, "outsidefill", 3); + + const results = await store.searchVec("ignored", "test-model", 3, embedded, undefined, query); + expect(results).toHaveLength(1); + expect(results[0]!.source).toBe("vec"); + expect(results[0]!.chunkPos).toBe(100); + expect(results[0]!.score).toBeCloseTo(0.6, 5); + + await cleanupTestDb(store); + }); + + test("searchVec union keeps a sibling collection reachable behind a long in-scope document", async () => { + const store = await createTestStore(); + const small = await createTestCollection({ name: "small", pwd: "/test/small" }); + const large = await createTestCollection({ name: "large", pwd: "/test/large" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, large, "longhash", Array.from({ length: 12 }, () => vector(1, 0))); + await insertVecDoc(store, small, "smallhash", [vector(0.6, 0.8)]); + const outside = await createTestCollection({ name: "outside", pwd: "/test/outside" }); + await insertFillerDocs(store, outside, "outsidefill", 4); + + const results = await store.searchVec("ignored", "test-model", 3, [small, large], undefined, query); + expect(results.map((r) => r.collectionName).sort()).toEqual([large, small].sort()); + expect(results.some((r) => r.hash === "smallhash")).toBe(true); + + await cleanupTestDb(store); + }); + + test("searchVec scoped to a collection still answers when one chunk row has no vector", async () => { + const store = await createTestStore(); + const partial = await createTestCollection({ name: "partial", pwd: "/test/partial" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, partial, "wholehash0", [vector(1, 0)]); + await insertVecDoc(store, partial, "wholehash1", [vector(0.9, 0.44)]); + await insertVecDoc(store, partial, "ghosthash", [vector(0.8, 0.6)]); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + await insertFillerDocs(store, other, "otherfill", 4); + const ghost = store.db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = 'ghosthash' AND seq = 0`).get() as { id: number }; + deletePartitionRows(store.db, [ghost.id]); + + const scoped = await store.searchVec("ignored", "test-model", 5, partial, undefined, query); + const unscoped = await store.searchVec("ignored", "test-model", 5, undefined, undefined, query); + expect(scoped.map((r) => r.hash)).toEqual(["wholehash0", "wholehash1"]); + expect(unscoped.filter((r) => r.collectionName === partial).map((r) => r.hash)).toEqual(["wholehash0", "wholehash1"]); + + await cleanupTestDb(store); + }); + + test("searchVec scoped to a collection that holds most of the index returns its nearest documents", async () => { + const store = await createTestStore(); + const big = await createTestCollection({ name: "big", pwd: "/test/big" }); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + store.ensureVecTable(DIMS); + for (let i = 0; i < 30; i++) { + await insertVecDoc(store, big, `bighash${String(i).padStart(2, "0")}`, [vector(1, 0.02 + i * 0.01)]); + } + for (let i = 0; i < 12; i++) { + await insertVecDoc(store, other, `otherhash${String(i).padStart(2, "0")}`, [vector(1, 0)]); + } + + const results = await store.searchVec("ignored", "test-model", 3, big, undefined, query); + expect(results.map((r) => r.hash)).toEqual(["bighash00", "bighash01", "bighash02"]); + + await cleanupTestDb(store); + }); + + test("searchVec scoped to a majority collection is not starved by nearer vectors outside it", async () => { + const store = await createTestStore(); + const big = await createTestCollection({ name: "big", pwd: "/test/big" }); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + store.ensureVecTable(DIMS); + for (let i = 0; i < 100; i++) { + await insertVecDoc(store, big, `bighash${String(i).padStart(3, "0")}`, [vector(0.6, 0.8)]); + } + for (let i = 0; i < 80; i++) { + await insertVecDoc(store, other, `otherhash${String(i).padStart(3, "0")}`, [vector(1, 0)]); + } + + const results = await store.searchVec("ignored", "test-model", 3, big, undefined, query); + expect(results).toHaveLength(3); + expect(results.every((r) => r.collectionName === big)).toBe(true); + + await cleanupTestDb(store); + }); + + test("searchVec scoped to a majority collection larger than the over-fetch returns its nearest documents", async () => { + const store = await createTestStore(); + const big = await createTestCollection({ name: "big", pwd: "/test/big" }); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + store.ensureVecTable(DIMS); + for (let i = 0; i < 300; i++) { + await insertVecDoc(store, big, `bighash${String(i).padStart(3, "0")}`, [vector(1, 0.02 + i * 0.001)]); + } + for (let i = 0; i < 20; i++) { + await insertVecDoc(store, other, `otherhash${String(i).padStart(2, "0")}`, [vector(1, 0)]); + } + + const results = await store.searchVec("ignored", "test-model", 3, big, undefined, query); + expect(results.map((r) => r.hash)).toEqual(["bighash000", "bighash001", "bighash002"]); + + await cleanupTestDb(store); + }); + + test("renameCollection keeps every vector reachable under the new name", async () => { + const store = await createTestStore(); + const before = await createTestCollection({ name: "before", pwd: "/test/before" }); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, before, "renamedhash", [vector(1, 0)]); + await insertFillerDocs(store, other, "otherfill", 3); + + renameCollection(store.db, before, "after"); + + const renamed = await store.searchVec("ignored", "test-model", 3, "after", undefined, query); + expect(renamed.map((r) => r.hash)).toEqual(["renamedhash"]); + expect(renamed[0]!.collectionName).toBe("after"); + expect(await store.searchVec("ignored", "test-model", 3, before, undefined, query)).toEqual([]); + expect(vectorRowCount(store, "after")).toBe(1); + + await cleanupTestDb(store); + }); + + test("removeCollection drops the collection's partition and leaves the others", async () => { + const store = await createTestStore(); + const gone = await createTestCollection({ name: "gone", pwd: "/test/gone" }); + const kept = await createTestCollection({ name: "kept", pwd: "/test/kept" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, gone, "gonehash", [vector(1, 0), vector(0.9, 0.1)]); + await insertVecDoc(store, kept, "kepthash", [vector(0.6, 0.8)]); + + removeCollection(store.db, gone); + + expect(vectorRowCount(store, gone)).toBe(0); + expect(resolveCollectionId(store.db, gone)).toBeUndefined(); + expect((store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get() as { c: number }).c).toBe(1); + const results = await store.searchVec("ignored", "test-model", 5, undefined, undefined, query); + expect(results.map((r) => r.hash)).toEqual(["kepthash"]); + + await cleanupTestDb(store); + }); + + test("clearAllEmbeddings for one collection keeps a shared hash in every collection", async () => { + const store = await createTestStore(); + const cleared = await createTestCollection({ name: "cleared", pwd: "/test/cleared" }); + const other = await createTestCollection({ name: "other", pwd: "/test/other" }); + store.ensureVecTable(DIMS); + await insertTestDocument(store.db, cleared, { name: "shared", hash: "sharedhash", body: "Document sharedhash", displayPath: "shared.md" }); + await insertTestDocument(store.db, other, { name: "shared", hash: "sharedhash", body: "Document sharedhash", displayPath: "shared.md" }); + insertChunkVectors(store, "sharedhash", [vector(1, 0)]); + await insertVecDoc(store, cleared, "onlyhash", [vector(0.9, 0.44)]); + + clearAllEmbeddings(store.db, cleared); + + expect((await store.searchVec("ignored", "test-model", 5, other, undefined, query)).map((r) => r.hash)).toEqual(["sharedhash"]); + expect((await store.searchVec("ignored", "test-model", 5, cleared, undefined, query)).map((r) => r.hash)).toEqual(["sharedhash"]); + expect(store.db.prepare(`SELECT COUNT(*) AS c FROM content_vectors WHERE hash = 'onlyhash'`).get()).toEqual({ c: 0 }); + expect(vectorRowCount(store, cleared)).toBe(1); + + await cleanupTestDb(store); + }); }); // ============================================================================= @@ -4515,7 +4910,7 @@ describe.skipIf(!!process.env.CI)("LlamaCpp Integration", () => { body: "Some content", }); - // No vectors_vec table exists, should return empty + // No vector table exists, should return empty const results = await store.searchVec("query", "embeddinggemma", 10); expect(results).toHaveLength(0); @@ -4538,8 +4933,7 @@ describe.skipIf(!!process.env.CI)("LlamaCpp Integration", () => { // Create vector table and insert a vector store.ensureVecTable(768); const embedding = Array(768).fill(0).map(() => Math.random()); - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(hash, new Date().toISOString()); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_0`, new Float32Array(embedding)); + store.insertEmbedding(hash, 0, 0, new Float32Array(embedding), 'test', new Date().toISOString()); const results = await store.searchVec("test query", "embeddinggemma", 10); expect(results).toHaveLength(1); @@ -4570,14 +4964,12 @@ describe.skipIf(!!process.env.CI)("LlamaCpp Integration", () => { body: "Content in collection two", }); - // Create vectors_vec table with correct dimensions (768 for embeddinggemma) + // Create the vector table with correct dimensions (768 for embeddinggemma) store.ensureVecTable(768); const embedding1 = Array(768).fill(0).map(() => Math.random()); const embedding2 = Array(768).fill(0).map(() => Math.random()); - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(hash1, new Date().toISOString()); - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(hash2, new Date().toISOString()); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash1}_0`, new Float32Array(embedding1)); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash2}_0`, new Float32Array(embedding2)); + store.insertEmbedding(hash1, 0, 0, new Float32Array(embedding1), 'test', new Date().toISOString()); + store.insertEmbedding(hash2, 0, 0, new Float32Array(embedding2), 'test', new Date().toISOString()); // Search without filter - should return both const allResults = await store.searchVec("content", "embeddinggemma", 10); @@ -4610,8 +5002,7 @@ describe.skipIf(!!process.env.CI)("LlamaCpp Integration", () => { // Create vector table and insert a test vector store.ensureVecTable(768); const embedding = Array(768).fill(0).map(() => Math.random()); - store.db.prepare(`INSERT INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, 0, 0, 'test', ?)`).run(hash, new Date().toISOString()); - store.db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_0`, new Float32Array(embedding)); + store.insertEmbedding(hash, 0, 0, new Float32Array(embedding), 'test', new Date().toISOString()); // This should complete quickly (not hang) due to the two-step fix // The old code with JOINs in the sqlite-vec query would hang indefinitely @@ -5294,7 +5685,7 @@ describe("Embedding batching", () => { expect(result.errors).toBeGreaterThan(0); expect(result.failures?.[0]?.attempts).toBe(3); expect(db.prepare(`SELECT COUNT(*) as count FROM content_vectors`).get()).toEqual({ count: 0 }); - expect(db.prepare(`SELECT COUNT(*) as count FROM vectors_vec`).get()).toEqual({ count: 0 }); + expect(db.prepare(`SELECT COUNT(*) as count FROM ${VEC_TABLE}`).get()).toEqual({ count: 0 }); expect(store.getHashesNeedingEmbedding()).toBe(1); expect(store.getStatus().needsEmbedding).toBe(1); } finally { @@ -5370,7 +5761,7 @@ describe("Embedding batching", () => { const db = store.db; // Store is pinned to a 3-dim embed model. Docs AND the query must use it, - // so vectors_vec is created as float[3]. + // so the vector table is created as float[3]. const storeModel = "hf:store/embeddinggemma-300M.gguf"; const storeLlm = { ...createFakeTokenizer(), @@ -5424,7 +5815,7 @@ describe("Embedding batching", () => { const model = "hf:Qwen/Qwen3-Embedding-0.6B-GGUF/Qwen3-Embedding-0.6B-Q8_0.gguf"; const searchVecSpy = vi.fn(async () => [] as SearchResult[]) as any; - store.db.exec(`CREATE TABLE vectors_vec (hash_seq TEXT PRIMARY KEY, embedding BLOB)`); + store.db.exec(`CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`); store.llm = { embedModelName: model } as any; store.searchVec = searchVecSpy as any; store.expandQuery = vi.fn(async () => []) as any; @@ -5450,7 +5841,7 @@ describe("Embedding batching", () => { }))); const searchVecSpy = vi.fn(async () => [] as SearchResult[]) as any; - store.db.exec(`CREATE TABLE vectors_vec (hash_seq TEXT PRIMARY KEY, embedding BLOB)`); + store.db.exec(`CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`); store.llm = { embedModelName: model, embedBatch: embedBatchSpy, @@ -5482,7 +5873,7 @@ describe("Embedding batching", () => { const searchVecSpy = vi.fn(async () => [] as SearchResult[]) as any; const searchFtsSpy = vi.fn(() => [] as SearchResult[]) as any; - store.db.exec(`CREATE TABLE vectors_vec (hash_seq TEXT PRIMARY KEY, embedding BLOB)`); + store.db.exec(`CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`); store.llm = { embedModelName: model, embedBatch: embedBatchSpy, @@ -5521,7 +5912,7 @@ describe("Embedding batching", () => { }))); const searchVecSpy = vi.fn(async () => [] as SearchResult[]) as any; - store.db.exec(`CREATE TABLE vectors_vec (hash_seq TEXT PRIMARY KEY, embedding BLOB)`); + store.db.exec(`CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`); store.llm = { embedModelName: model, embedBatch: embedBatchSpy, From d44eeb9cea3e86e11169d658d82077d3e2c96f88 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 3 Sep 2026 13:18:52 -0500 Subject: [PATCH 31/82] feat(embed): copy vectors to collections that gain an embedded hash and write each batch in one transaction `generateEmbeddings` starts with a copy pass: every active (hash, collection) pair whose chunks are in content_vectors but missing from that collection's partition gets its rows copied from whichever partition holds them, so a document that appears in a second collection is searchable there after the next embed run without another model call. A chunk no partition holds has its hash's rows removed instead, which puts the hash back in front of the pending detector; that closes the case where content_vectors claims a chunk the index lost. `EmbedResult.chunksCopied` reports the copies and `qmd embed` prints them. Each 32-chunk batch is written inside one IMMEDIATE transaction, with the success and failure bookkeeping applied after the commit. A chunk writes several rows across three tables, and a kill or a write error mid-batch now rolls the whole batch back instead of leaving content_vectors ahead of the index; the existing per-chunk fallback then retries the batch's chunks one by one. (cherry picked from commit 48e564fd1bee4feb6ac5a026eb9f900a8a3d48be) --- src/cli/qmd.ts | 3 + src/store.ts | 79 ++++++++++++++++++++++---- test/store.test.ts | 137 +++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 207 insertions(+), 12 deletions(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index db51bc794..ffd8e5626 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -2452,6 +2452,9 @@ async function vectorIndex( const totalTimeSec = result.durationMs / 1000; + if (result.chunksCopied > 0) { + console.log(`${c.green}✓${c.reset} Copied ${formatCount(result.chunksCopied)} vectors into collections that gained already-embedded documents`); + } if (result.chunksEmbedded === 0 && result.docsProcessed === 0) { console.log(`${c.green}✓ No non-empty documents to embed.${c.reset}`); } else { diff --git a/src/store.ts b/src/store.ts index 2d0cb06e1..81266ce88 100644 --- a/src/store.ts +++ b/src/store.ts @@ -13,6 +13,7 @@ import { openDatabase, loadSqliteVec } from "./db.js"; import { + PartitionWriter, VEC_COLLECTION_IDS_TABLE, VEC_ROWS_TABLE, VEC_TABLE, @@ -22,6 +23,7 @@ import { deleteCollectionId, deletePartitionRows, hasVectorIndex, + missingPartitionRows, partitionRowKey, renameCollectionId, resolveCollectionId, @@ -2062,6 +2064,8 @@ export type EmbedProgress = { export type EmbedResult = { docsProcessed: number; chunksEmbedded: number; + /** Chunks copied into the partition of a collection that gained an already-embedded hash. */ + chunksCopied: number; /** Active failed chunks that did not recover after retries. */ errors: number; failures?: EmbedFailure[]; @@ -2305,10 +2309,11 @@ export async function generateEmbeddings( clearAllEmbeddings(db, options?.collection); } + const chunksCopied = copyVectorsToNewCollections(db, options?.collection).copied; const docsToEmbed = getPendingEmbeddingDocs(db, options?.collection, model); if (docsToEmbed.length === 0) { - return { docsProcessed: 0, chunksEmbedded: 0, errors: 0, durationMs: 0 }; + return { docsProcessed: 0, chunksEmbedded: 0, chunksCopied, errors: 0, durationMs: 0 }; } const totalBytes = docsToEmbed.reduce((sum, doc) => sum + Math.max(0, doc.bytes), 0); const totalDocs = docsToEmbed.length; @@ -2484,19 +2489,30 @@ export async function generateEmbeddings( try { const embeddings = await session.embedBatch(texts, { model }); - for (let i = 0; i < chunkBatch.length; i++) { - const chunk = chunkBatch[i]!; - const embedding = embeddings[i]; - if (embedding) { - insertEmbedding(db, chunk.hash, chunk.seq, chunk.pos, new Float32Array(embedding.embedding), model, now, chunk.expectedTotalChunks, fingerprint); - chunksEmbedded++; - successesSinceRetry++; - clearFailure(chunk); - } else { - recordFailure(chunk, "batch embedding returned no vector"); + // One IMMEDIATE transaction per batch: every chunk writes several + // rows across three tables, and a kill mid-batch must not leave + // them out of step. Failure bookkeeping runs after the commit. + const stored: ChunkItem[] = []; + const unembedded: ChunkItem[] = []; + db.transaction(() => { + for (let i = 0; i < chunkBatch.length; i++) { + const chunk = chunkBatch[i]!; + const embedding = embeddings[i]; + if (embedding) { + insertEmbedding(db, chunk.hash, chunk.seq, chunk.pos, new Float32Array(embedding.embedding), model, now, chunk.expectedTotalChunks, fingerprint); + stored.push(chunk); + } else { + unembedded.push(chunk); + } } - batchChunkBytesProcessed += chunk.bytes; + }).immediate(); + for (const chunk of stored) { + chunksEmbedded++; + successesSinceRetry++; + clearFailure(chunk); } + for (const chunk of unembedded) recordFailure(chunk, "batch embedding returned no vector"); + batchChunkBytesProcessed += chunkBatch.reduce((sum, chunk) => sum + chunk.bytes, 0); await retryFailedChunks(); } catch (error) { // Batch failed — try individual embeddings as fallback. If an @@ -2545,6 +2561,7 @@ export async function generateEmbeddings( return { docsProcessed: totalDocs, chunksEmbedded: result.chunksEmbedded, + chunksCopied, errors: result.errors, failures: result.failures, durationMs: Date.now() - startTime, @@ -4864,6 +4881,44 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { }); } +/** + * Give every active (hash, collection) pair whose chunks are embedded a row + * in that collection's partition, copied from any partition that holds the + * chunk, so a hash that gains a collection is searchable there without + * another model call. A chunk no partition holds is queued for embedding by + * deleting its hash's rows, which the pending detector then picks up. + */ +export function copyVectorsToNewCollections(db: Database, collection?: string): { copied: number; queued: number } { + const layout = vecLayout(db); + if (layout.kind === "legacy" || !isSqliteVecAvailable()) return { copied: 0, queued: 0 }; + return withLazyContentVectorMigration(db, () => db.transaction(() => { + const missing = missingPartitionRows(db, collection); + if (missing.length === 0) return { copied: 0, queued: 0 }; + const writer = layout.kind === "partitioned" ? new PartitionWriter(db) : null; + const sourceOf = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? LIMIT 1`); + const vectorOf = writer ? db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`) : null; + const queued = new Set(); + let copied = 0; + for (const row of missing) { + if (queued.has(row.hash)) continue; + const source = sourceOf.get(row.hash, row.seq) as { id: number } | undefined; + const vector = source && vectorOf ? (vectorOf.get(vecInteger(source.id)) as { embedding: Uint8Array } | undefined) : undefined; + if (!vector || !writer) { + queued.add(row.hash); + continue; + } + if (writer.writeToCollection(row.hash, row.seq, row.collection, vector.embedding)) copied++; + } + const partitionRowsOf = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ?`); + const deleteChunks = db.prepare(`DELETE FROM content_vectors WHERE hash = ?`); + for (const hash of queued) { + deletePartitionRows(db, (partitionRowsOf.all(hash) as { id: number }[]).map((row) => row.id)); + deleteChunks.run(hash); + } + return { copied, queued: queued.size }; + }).immediate()); +} + /** * Insert a single embedding: the content_vectors row plus one row in the * partition of every active collection that holds the hash, in one diff --git a/test/store.test.ts b/test/store.test.ts index 0a0f92856..8e056056a 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -5936,6 +5936,143 @@ describe("Embedding batching", () => { } }); + test("generateEmbeddings copies vectors to a collection that gained an embedded hash instead of embedding again", async () => { + const store = await createTestStore(); + const db = store.db; + const fakeLlm = createFakeEmbedLlm(); + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + + try { + const first = await createTestCollection({ name: "first", pwd: "/test/first" }); + const second = await createTestCollection({ name: "second", pwd: "/test/second" }); + await insertTestDocument(db, first, { name: "shared", hash: "sharedhash", body: "# Shared\n\nShared body", displayPath: "shared.md" }); + const embedded = await generateEmbeddings(store); + expect(embedded.chunksEmbedded).toBe(1); + expect(embedded.chunksCopied).toBe(0); + + await insertTestDocument(db, second, { name: "copy", hash: "sharedhash", body: "# Shared\n\nShared body", displayPath: "copy.md" }); + expect(await store.searchVec("ignored", "test-model", 5, second, undefined, [1, 2, 3])).toEqual([]); + expect(store.getHashesNeedingEmbedding()).toBe(0); + + const result = await generateEmbeddings(store); + + expect(fakeLlm.embedBatchCalls).toHaveLength(1); + expect(result.chunksCopied).toBe(1); + expect(result.chunksEmbedded).toBe(0); + const found = await store.searchVec("ignored", "test-model", 5, second, undefined, [1, 2, 3]); + expect(found.map((r) => r.displayPath)).toEqual([`${second}/copy.md`]); + expect((await store.searchVec("ignored", "test-model", 5, first, undefined, [1, 2, 3])).map((r) => r.displayPath)).toEqual([`${first}/shared.md`]); + } finally { + setDefaultLlamaCpp(null); + await cleanupTestDb(store); + } + }); + + test("generateEmbeddings embeds a hash again when no partition holds its vector", async () => { + const store = await createTestStore(); + const db = store.db; + const fakeLlm = createFakeEmbedLlm(); + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + + try { + const docs = await createTestCollection({ name: "docs", pwd: "/test/docs" }); + await insertTestDocument(db, docs, { name: "lost", hash: "losthash", body: "# Lost\n\nLost body", displayPath: "lost.md" }); + await generateEmbeddings(store); + const rows = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE}`).all() as { id: number }[]; + expect(rows).toHaveLength(1); + deletePartitionRows(db, rows.map((row) => row.id)); + expect(store.getHashesNeedingEmbedding()).toBe(0); + + const result = await generateEmbeddings(store); + + expect(fakeLlm.embedBatchCalls).toHaveLength(2); + expect(result.chunksEmbedded).toBe(1); + expect(result.chunksCopied).toBe(0); + expect((await store.searchVec("ignored", "test-model", 5, docs, undefined, [1, 2, 3])).map((r) => r.hash)).toEqual(["losthash"]); + } finally { + setDefaultLlamaCpp(null); + await cleanupTestDb(store); + } + }); + + test("generateEmbeddings writes a batch in one transaction and rolls it back when a chunk write fails", async () => { + const store = await createTestStore(); + const real = store.db; + const fakeLlm = createFakeEmbedLlm(); + const doomed = "doomedhash"; + const observed = { inserted: [] as string[], insideTransaction: [] as boolean[], firstHashRowsAfterRollback: -1 }; + const rowsOf = (hash: string) => (real.prepare(`SELECT COUNT(*) AS c FROM content_vectors WHERE hash = ?`).get(hash) as { c: number }).c; + const guard = unknown>(run: T): T => ((...args: unknown[]) => { + try { + return run(...args); + } catch (error) { + // The failing write unwinds an inner savepoint first; sample once the + // outermost transaction has rolled back. + if (!(real as any).inTransaction && observed.firstHashRowsAfterRollback < 0 && observed.inserted.length > 0) { + observed.firstHashRowsAfterRollback = rowsOf(observed.inserted[0]!); + } + throw error; + } + }) as T; + const failingDb = { + prepare: (sql: string) => { + const stmt = real.prepare(sql); + if (!sql.includes("INSERT OR REPLACE INTO content_vectors")) return stmt; + return { + run: (...params: any[]) => { + if (params[0] === doomed) throw new Error("injected write failure"); + observed.inserted.push(params[0]); + observed.insideTransaction.push((real as any).inTransaction); + return stmt.run(...params); + }, + get: (...params: any[]) => stmt.get(...params), + all: (...params: any[]) => stmt.all(...params), + iterate: (...params: any[]) => stmt.iterate(...params), + }; + }, + transaction: (fn: any) => { + const tx = real.transaction(fn); + const wrapped = guard(tx) as any; + wrapped.immediate = guard(tx.immediate); + return wrapped; + }, + exec: (sql: string) => real.exec(sql), + loadExtension: (path: string) => real.loadExtension(path), + close: () => real.close(), + }; + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + store.db = failingDb as any; + + try { + await insertTestDocument(real, "docs", { name: "one", hash: "onehash", body: "# One\n\nAlpha" }); + await insertTestDocument(real, "docs", { name: "two", hash: doomed, body: "# Two\n\nBeta" }); + await insertTestDocument(real, "docs", { name: "three", hash: "threehash", body: "# Three\n\nGamma" }); + + const result = await generateEmbeddings(store); + + expect(fakeLlm.embedBatchCalls).toHaveLength(1); + expect(observed.inserted[0]).toBe("onehash"); + expect(observed.insideTransaction.every(Boolean)).toBe(true); + expect(observed.firstHashRowsAfterRollback).toBe(0); + expect(result.errors).toBe(1); + expect(result.failures?.[0]?.hash).toBe(doomed); + expect(rowsOf(doomed)).toBe(0); + expect(rowsOf("onehash")).toBe(1); + expect(rowsOf("threehash")).toBe(1); + expect(store.getHashesNeedingEmbedding()).toBe(1); + } finally { + store.db = real; + setDefaultLlamaCpp(null); + await cleanupTestDb(store); + } + }); + test("generateEmbeddings rejects invalid batch limits", async () => { const store = await createTestStore(); From 51eae40c8febb6d7b6684fd12f007dd56d9abb46 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 3 Sep 2026 13:25:28 -0500 Subject: [PATCH 32/82] feat(update): remove stale vector rows at the end of every qmd update A changed file keeps its document row and gets a new hash, which strands the old hash's rows in the collection's vector partition. The partition filter cannot see `documents.active`, so on a churning collection those rows take k slots from every scoped search until something removes them; with enough of them a scoped query comes back short or empty. `qmd update` now runs the orphan-vector cleanup as its last step and reports the rows it removed, and the hint that used to suggest `qmd cleanup` for orphaned chunks goes away with the orphans. The regression test re-indexes a file through four rewrites with a fake embedder that places the stale bodies nearer the query than the live documents, shows the scoped search returning nothing, and checks that the cleanup restores the full count. Two tests cover the legacy-fingerprint adoption sample, which resolves its nearest stored vector through the partition mapping. (cherry picked from commit 17589bc31e4510849336b400da1d2124e31d4254) --- src/cli/qmd.ts | 23 ++++------ test/cli.test.ts | 43 ++++++++++++++++-- test/store.test.ts | 111 ++++++++++++++++++++++++++++++++++++++++++++- 3 files changed, 160 insertions(+), 17 deletions(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index ffd8e5626..c9a7202df 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -57,6 +57,7 @@ import { deactivateDocument, getActiveDocumentPaths, cleanupOrphanedContent, + cleanupOrphanedVectors, countOrphanedVectors, previewCleanup, runCleanup, @@ -526,14 +527,6 @@ function sanitizeDiagnosticMessage(message: string): string { .join("; "); } -/** Hint after `qmd update` when orphaned embedding chunks exceed this share of vectors (#768). */ -const ORPHAN_VECTOR_HINT_RATIO = 0.1; - -function formatOrphanedVectorHint(orphaned: number, total: number): string { - const pct = total > 0 ? Math.round((orphaned / total) * 100) : 0; - return `${orphaned} orphaned embedding chunks (${pct}% of vectors) — run 'qmd cleanup' to reclaim space`; -} - async function showStatus(): Promise { const dbPath = getDbPath(); const db = getDb(); @@ -1023,19 +1016,23 @@ async function updateCollections(): Promise { console.log(""); } + // A changed file rewrites its document's hash in place, which strands the + // old hash's rows in the collection's vector partition; the partition + // filter cannot see documents.active, so those rows would take k slots + // from a scoped search until they are removed. + const staleVectors = cleanupOrphanedVectors(db); + // Check if any documents need embedding (show once at end) const needsEmbedding = getHashesNeedingEmbedding(db); - const vectorTotal = (db.prepare(`SELECT COUNT(*) as count FROM content_vectors`).get() as { count: number }).count; - const orphanedVectors = countOrphanedVectors(db); closeDb(); console.log(`${c.green}✓ All collections updated.${c.reset}`); + if (staleVectors > 0) { + console.log(`Removed ${staleVectors} stale vector row(s)`); + } if (needsEmbedding > 0) { console.log(`\nRun 'qmd embed' to update embeddings (${needsEmbedding} unique hashes need vectors)`); } - if (vectorTotal > 0 && orphanedVectors / vectorTotal >= ORPHAN_VECTOR_HINT_RATIO) { - console.log(`\n${formatOrphanedVectorHint(orphanedVectors, vectorTotal)}`); - } } /** diff --git a/test/cli.test.ts b/test/cli.test.ts index 04aec8e27..18959f9b5 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -15,6 +15,8 @@ import { spawn } from "child_process"; import { setTimeout as sleep } from "timers/promises"; import { buildEditorUri, termLink, resolveEmbedModelForCli } from "../src/cli/qmd.ts"; import { openDatabase } from "../src/db.ts"; +import { createStore, insertContent, insertDocument } from "../src/store.ts"; +import { VEC_ROWS_TABLE, VEC_TABLE } from "../src/vec-layout.ts"; import { DEFAULT_EMBED_MODEL_URI, DEFAULT_GENERATE_MODEL_URI, DEFAULT_RERANK_MODEL_URI } from "../src/llm.ts"; import { setConfigSource } from "../src/collections.ts"; @@ -1369,6 +1371,39 @@ describe("CLI Cleanup Command", () => { }); }); +describe("qmd update stale vector rows", () => { + test("a document whose file disappeared loses its vector rows at the end of update", async () => { + const env = await createIsolatedTestEnv("stale-vector-rows"); + const add = await runQmd(["collection", "add", fixturesDir, "--name", "fixtures"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(add.exitCode).toBe(0); + + const store = createStore(env.dbPath); + try { + const now = new Date().toISOString(); + insertContent(store.db, "ghosthash", "# Ghost\n\ngone", now); + insertDocument(store.db, "fixtures", "ghost.md", "Ghost", "ghosthash", now, now); + store.ensureVecTable(3); + store.insertEmbedding("ghosthash", 0, 0, new Float32Array([1, 2, 3]), "test", now); + expect(store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get()).toEqual({ c: 1 }); + } finally { + store.close(); + } + + const update = await runQmd(["update"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(update.exitCode).toBe(0); + expect(update.stdout).toContain("Removed 1 stale vector row(s)"); + + const db = openDatabase(env.dbPath); + try { + expect(db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get()).toEqual({ c: 0 }); + expect(db.prepare(`SELECT COUNT(*) AS c FROM content_vectors WHERE hash = 'ghosthash'`).get()).toEqual({ c: 0 }); + expect(db.prepare(`SELECT active FROM documents WHERE path = 'ghost.md'`).get()).toEqual({ active: 0 }); + } finally { + db.close(); + } + }); +}); + describe("orphaned embedding vectors (#768)", () => { let localDbPath: string; let localConfigDir: string; @@ -1429,11 +1464,13 @@ describe("orphaned embedding vectors (#768)", () => { expect(stdout).toContain("qmd cleanup"); }); - test("update hints when orphan ratio exceeds 10%", async () => { + test("update leaves orphaned chunks for cleanup when there is no vector table", async () => { const { stdout, exitCode } = await runQmd(["update"], { dbPath: localDbPath, configDir: localConfigDir }); expect(exitCode).toBe(0); - expect(stdout).toContain("3 orphaned embedding chunks (75% of vectors)"); - expect(stdout).toContain("run 'qmd cleanup' to reclaim space"); + expect(stdout).not.toContain("orphaned embedding chunks"); + expect(stdout).not.toContain("stale vector row"); + const status = await runQmd(["status"], { dbPath: localDbPath, configDir: localConfigDir }); + expect(status.stdout).toMatch(/Orphaned:\s+3 embedding chunks/); }); test("cleanup --dry-run reports what would be removed without deleting", async () => { diff --git a/test/store.test.ts b/test/store.test.ts index 8e056056a..2e7ed22ce 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -59,10 +59,10 @@ import { insertDocument, cleanupOrphanedVectors, clearAllEmbeddings, + maybeAdoptLegacyEmbeddingFingerprint, removeCollection, renameCollection, generateEmbeddings, - maybeAdoptLegacyEmbeddingFingerprint, getHybridRrfWeights, _resetProductionModeForTesting, hybridQuery, @@ -6073,6 +6073,115 @@ describe("Embedding batching", () => { } }); + test("cleanupOrphanedVectors after re-indexing a changed file removes the stale rows that starve a scoped search", async () => { + const store = await createTestStore(); + const db = store.db; + const near = [1, 0, 0]; + const far = [0, 1, 0]; + const embeddingFor = (text: string) => ({ embedding: text.includes("stale") ? near : far, model: "fake-embed" }); + const fakeLlm = { + ...createFakeTokenizer(), + async embed(text: string, _options?: { model?: string }) { return embeddingFor(text); }, + async embedBatch(texts: string[], _options?: { model?: string }) { return texts.map(embeddingFor); }, + }; + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + const dir = await mkdtemp(join(tmpdir(), "qmd-stale-vectors-")); + + try { + await writeFile(join(dir, "live.md"), "# Live\n\nlive body\n"); + const versions = ["stale 0", "stale 1", "stale 2", "stale 3", "final body"]; + for (const body of versions) { + await writeFile(join(dir, "churn.md"), `# Churn\n\n${body}\n`); + await reindexCollection(store, dir, "**/*.md", "docs"); + await generateEmbeddings(store); + } + expect(db.prepare(`SELECT COUNT(*) AS c FROM documents WHERE active = 1`).get()).toEqual({ c: 2 }); + expect(db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get()).toEqual({ c: 6 }); + + // Four stale rows sit nearer the query than both live documents: with + // limit 1 every one of the scoped KNN's three slots goes to a row no + // active document owns, and the search comes back empty. + expect(await store.searchVec("ignored", "test-model", 1, "docs", undefined, near)).toEqual([]); + + expect(cleanupOrphanedVectors(db)).toBe(4); + + expect(db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get()).toEqual({ c: 2 }); + const after = await store.searchVec("ignored", "test-model", 1, "docs", undefined, near); + expect(after).toHaveLength(1); + expect(await store.searchVec("ignored", "test-model", 5, "docs", undefined, near)).toHaveLength(2); + } finally { + setDefaultLlamaCpp(null); + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("maybeAdoptLegacyEmbeddingFingerprint finds the sample's stored vector through the partition mapping", async () => { + const store = await createTestStore(); + const db = store.db; + const model = "hf:test/embed-model.gguf"; + const stored = [0.1, 0.2, 0.3]; + const fakeLlm = { + ...createFakeTokenizer(), + embedModelName: model, + async embed(_text: string, _options?: { model?: string }) { return { embedding: stored, model }; }, + async embedBatch(texts: string[], _options?: { model?: string }) { return texts.map(() => ({ embedding: stored, model })); }, + }; + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + + try { + const docs = await createTestCollection({ name: "docs", pwd: "/test/docs" }); + await insertTestDocument(db, docs, { name: "legacy", hash: "legacyhash", body: "# Legacy\n\nLegacy body", displayPath: "legacy.md" }); + store.ensureVecTable(3); + store.insertEmbedding("legacyhash", 0, 0, new Float32Array(stored), model, new Date().toISOString(), 1, ""); + expect(store.getHashesNeedingEmbedding(model)).toBe(1); + + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, model); + + expect(result).toMatchObject({ checked: true, adopted: 1 }); + expect(result.reason).toContain("legacyhash_0"); + expect(store.getHashesNeedingEmbedding(model)).toBe(0); + } finally { + setDefaultLlamaCpp(null); + await cleanupTestDb(store); + } + }); + + test("maybeAdoptLegacyEmbeddingFingerprint keeps a legacy fingerprint whose sample no longer matches", async () => { + const store = await createTestStore(); + const db = store.db; + const model = "hf:test/embed-model.gguf"; + const fakeLlm = { + ...createFakeTokenizer(), + embedModelName: model, + async embed(_text: string, _options?: { model?: string }) { return { embedding: [0.3, 0.2, 0.1], model }; }, + async embedBatch(texts: string[], _options?: { model?: string }) { return texts.map(() => ({ embedding: [0.3, 0.2, 0.1], model })); }, + }; + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + + try { + const docs = await createTestCollection({ name: "docs", pwd: "/test/docs" }); + await insertTestDocument(db, docs, { name: "legacy", hash: "legacyhash", body: "# Legacy\n\nLegacy body", displayPath: "legacy.md" }); + store.ensureVecTable(3); + store.insertEmbedding("legacyhash", 0, 0, new Float32Array([0.1, 0.2, 0.3]), model, new Date().toISOString(), 1, ""); + + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, model); + + expect(result).toMatchObject({ checked: true, adopted: 0 }); + expect(result.reason).toContain("nearest legacyhash_0"); + expect(store.getHashesNeedingEmbedding(model)).toBe(1); + } finally { + setDefaultLlamaCpp(null); + await cleanupTestDb(store); + } + }); + test("generateEmbeddings rejects invalid batch limits", async () => { const store = await createTestStore(); From 23d5a6c1b327f0931ba4bac17799e7ce093976b7 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 3 Sep 2026 13:25:28 -0500 Subject: [PATCH 33/82] docs: describe the per-collection vector index layout CHANGELOG gets the Unreleased entry for the index format change: what the partitioned layout does for scoped queries, that the first command after upgrading converts the index in place with progress, resumption and shared work between concurrent commands, what rename, update and embed now do with vectors, and that downgrading needs `qmd embed -f` and a running `qmd mcp` server a restart. The README schema block lists `vector_collection_ids`, `vector_rows` and `vectors_by_collection` in place of `vectors_vec`. (cherry picked from commit c9cfc20e40864ac182f6dddb0f5cfaf0c604d07d) --- CHANGELOG.md | 17 +++++++++++++++++ README.md | 16 +++++++++------- 2 files changed, 26 insertions(+), 7 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 34488839f..82cd71288 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -86,6 +86,23 @@ ranked list per search. Rankings no longer depend on the order the collections are named. #946 (thanks @shalom-t), #1009 (thanks @xidus90) +### Changed + +- The vector index is partitioned by collection. Each embedded chunk is + stored once per collection that holds it, under an integer collection id, + so a collection-scoped `qmd query` or `qmd vsearch` scans only that + collection's vectors instead of pre-filtering the whole index, and a small + collection is never crowded out of the results by a large one. The first + command after upgrading converts the index in place and prints its + progress; the conversion resumes from where it stopped if interrupted, two + commands started at once share it, and a VACUUM at the end reclaims the + space of the old table. `qmd collection rename` no longer touches vectors, + `qmd update` removes the vector rows of changed files as it finishes, and + `qmd embed` copies an already-embedded document into a collection that + gains it instead of embedding it again. Downgrading to an older qmd + afterwards needs `qmd embed -f`, and a `qmd mcp` server started before the + upgrade must be restarted. + ## [2.8.3] - 2026-08-16 ### Security diff --git a/README.md b/README.md index d1b50c396..d6ee6ed0e 100644 --- a/README.md +++ b/README.md @@ -1415,13 +1415,15 @@ Index stored in: `~/.cache/qmd/index.sqlite` ### Schema ```sql -collections -- Indexed directories with name and glob patterns -path_contexts -- Context descriptions by virtual path (qmd://...) -documents -- Markdown content with metadata and docid (6-char hash) -documents_fts -- FTS5 full-text index -content_vectors -- Embedding chunks (hash, seq, pos, 900 tokens each) -vectors_vec -- sqlite-vec vector index (hash_seq key) -llm_cache -- Cached LLM responses (query expansion, rerank scores) +collections -- Indexed directories with name and glob patterns +path_contexts -- Context descriptions by virtual path (qmd://...) +documents -- Markdown content with metadata and docid (6-char hash) +documents_fts -- FTS5 full-text index +content_vectors -- Embedding chunks (hash, seq, pos, 900 tokens each) +vector_collection_ids -- Integer id per collection name (the vector partition key) +vector_rows -- Vector rowid to (hash, seq, collection_id) +vectors_by_collection -- sqlite-vec vector index, one row per chunk and collection +llm_cache -- Cached LLM responses (query expansion, rerank scores) ``` ## Environment Variables From 530a6fd4ed112e8a8f0af032151f4cec4f77c58a Mon Sep 17 00:00:00 2001 From: Brett Date: Fri, 4 Sep 2026 16:11:17 -0500 Subject: [PATCH 34/82] refactor(store): share the partition row helpers across the embed, copy and clear paths `activeCollectionsOfHash`, `deletePartitionRowsOfHash` and a prepared-statement `storedEmbeddingLookup` in vec-layout.ts replace three copies of the same SQL in store.ts, the migration reuses the exported `parseDimensions` instead of its own regex, and the unused `STORE_SCHEMA_VERSION` alias goes. The collection-scoped `clearAllEmbeddings` now runs inside one IMMEDIATE transaction, so a force re-embed of one collection is a single commit and a crash cannot leave the mapping and the vec0 table out of step. (cherry picked from commit 459ef0c1148e85e0e662c293a6152fc9c71c91ef) --- src/cli/embed-lock.ts | 6 +++--- src/store-migrations.ts | 5 ++--- src/store.ts | 24 +++++++++++------------- src/vec-layout.ts | 37 +++++++++++++++++++++++++++++++------ 4 files changed, 47 insertions(+), 25 deletions(-) diff --git a/src/cli/embed-lock.ts b/src/cli/embed-lock.ts index e05ffac2f..5fdf9f2db 100644 --- a/src/cli/embed-lock.ts +++ b/src/cli/embed-lock.ts @@ -3,9 +3,9 @@ * * Concurrent embed runs against the same index can race on vector_rows * (UNIQUE constraint on hash, seq, collection_id). This lockfile keeps a - * second process from - * starting while another embed holds the lock. Stale files left by crashed - * processes are recovered via PID identity checks (same spirit as mcp-pid.ts). + * second process from starting while another embed holds the lock. Stale + * files left by crashed processes are recovered via PID identity checks + * (same spirit as mcp-pid.ts). */ import { existsSync, readFileSync, unlinkSync, writeFileSync } from "node:fs"; diff --git a/src/store-migrations.ts b/src/store-migrations.ts index ead3635d1..172aac097 100644 --- a/src/store-migrations.ts +++ b/src/store-migrations.ts @@ -22,6 +22,7 @@ import { VEC_TABLE, createPartitionedVecTable, missingPartitionRows, + parseDimensions, vecLayout, vecTableReadable, type ReadableVecLayout, @@ -29,7 +30,6 @@ import { export const FTS_SYNC_TRIGGERS_VERSION = 1; export const VECTOR_PARTITION_VERSION = 2; -export const STORE_SCHEMA_VERSION = VECTOR_PARTITION_VERSION; const CURSOR_KEY = "vector_partition_cursor"; @@ -193,8 +193,7 @@ function ensurePartitionedTable(db: Database, dimensions: number): void { createPartitionedVecTable(db, dimensions); return; } - const match = existing.sql.match(/float\[(\d+)\]/); - const existingDims = match?.[1] ? parseInt(match[1], 10) : null; + const existingDims = parseDimensions(existing.sql); if (existingDims !== dimensions) { throw new Error( `Vector migration found a ${VEC_TABLE} table with ${existingDims ?? "unknown"} dimensions while the legacy ${layout.kind} table holds ${dimensions}d vectors` diff --git a/src/store.ts b/src/store.ts index 81266ce88..2203b2a18 100644 --- a/src/store.ts +++ b/src/store.ts @@ -17,11 +17,13 @@ import { VEC_COLLECTION_IDS_TABLE, VEC_ROWS_TABLE, VEC_TABLE, + activeCollectionsOfHash, allocateCollectionId, createPartitionedVecTable, createVectorMetadataTables, deleteCollectionId, deletePartitionRows, + deletePartitionRowsOfHash, hasVectorIndex, missingPartitionRows, partitionRowKey, @@ -29,6 +31,7 @@ import { resolveCollectionId, resolveCollectionIds, rowidList, + storedEmbeddingLookup, upsertPartitionVector, vecInteger, vecLayout, @@ -4857,7 +4860,7 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { `; const collectionId = resolveCollectionId(db, collection); - withLazyContentVectorMigration(db, () => { + withLazyContentVectorMigration(db, () => db.transaction(() => { if (collectionId !== undefined) { const rows = db.prepare(` SELECT id FROM ${VEC_ROWS_TABLE} @@ -4878,7 +4881,7 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); } - }); + }).immediate()); } /** @@ -4895,24 +4898,21 @@ export function copyVectorsToNewCollections(db: Database, collection?: string): const missing = missingPartitionRows(db, collection); if (missing.length === 0) return { copied: 0, queued: 0 }; const writer = layout.kind === "partitioned" ? new PartitionWriter(db) : null; - const sourceOf = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? LIMIT 1`); - const vectorOf = writer ? db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`) : null; + const storedVector = writer ? storedEmbeddingLookup(db) : null; const queued = new Set(); let copied = 0; for (const row of missing) { if (queued.has(row.hash)) continue; - const source = sourceOf.get(row.hash, row.seq) as { id: number } | undefined; - const vector = source && vectorOf ? (vectorOf.get(vecInteger(source.id)) as { embedding: Uint8Array } | undefined) : undefined; + const vector = storedVector?.(row.hash, row.seq); if (!vector || !writer) { queued.add(row.hash); continue; } - if (writer.writeToCollection(row.hash, row.seq, row.collection, vector.embedding)) copied++; + if (writer.writeToCollection(row.hash, row.seq, row.collection, vector)) copied++; } - const partitionRowsOf = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ?`); const deleteChunks = db.prepare(`DELETE FROM content_vectors WHERE hash = ?`); for (const hash of queued) { - deletePartitionRows(db, (partitionRowsOf.all(hash) as { id: number }[]).map((row) => row.id)); + deletePartitionRowsOfHash(db, hash); deleteChunks.run(hash); } return { copied, queued: queued.size }; @@ -4940,8 +4940,7 @@ export function insertEmbedding( db.transaction(() => { db.prepare(`INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at) VALUES (?, ?, ?, ?, ?, ?, ?)`) .run(hash, seq, pos, model, fingerprint, totalChunks, embeddedAt); - const collections = db.prepare(`SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`).all(hash) as { collection: string }[]; - for (const { collection } of collections) { + for (const collection of activeCollectionsOfHash(db, hash)) { upsertPartitionVector(db, hash, seq, allocateCollectionId(db, collection), embedding); } })(); @@ -4953,13 +4952,12 @@ function removeIncompleteEmbeddings(db: Database, expectedChunksByHash: Map row.id)); + deletePartitionRowsOfHash(db, hash); deleteContentStmt.run(hash, model); removed += rows.length; } diff --git a/src/vec-layout.ts b/src/vec-layout.ts index ba3328c81..e28ec16c1 100644 --- a/src/vec-layout.ts +++ b/src/vec-layout.ts @@ -45,7 +45,7 @@ function shadowTables(table: string): VecShadowTables { }; } -function parseDimensions(sql: string | null): number | null { +export function parseDimensions(sql: string | null): number | null { const match = sql?.match(/float\[(\d+)\]/); return match?.[1] ? parseInt(match[1], 10) : null; } @@ -186,6 +186,19 @@ export function deletePartitionRows(db: Database, rowids: readonly number[]): vo } } +/** Deletes every partition row of a hash, in every collection. */ +export function deletePartitionRowsOfHash(db: Database, hash: string): void { + const rows = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ?`).all(hash) as { id: number }[]; + deletePartitionRows(db, rows.map((row) => row.id)); +} + +const ACTIVE_COLLECTIONS_OF_HASH_SQL = `SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`; + +/** Names of the collections with an active document for the hash: the partitions that must hold its vectors. */ +export function activeCollectionsOfHash(db: Database, hash: string): string[] { + return (db.prepare(ACTIVE_COLLECTIONS_OF_HASH_SQL).all(hash) as { collection: string }[]).map((row) => row.collection); +} + /** * Replace or insert the vector of (hash, seq) in one collection's partition. * vec0 ignores OR REPLACE, so an existing row is deleted and its rowid reused. @@ -211,7 +224,7 @@ export class PartitionWriter { private readonly insertVec; constructor(private readonly db: Database) { - this.collectionsOf = db.prepare(`SELECT DISTINCT collection FROM documents WHERE hash = ? AND active = 1`); + this.collectionsOf = db.prepare(ACTIVE_COLLECTIONS_OF_HASH_SQL); this.insertRow = db.prepare(`INSERT OR IGNORE INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES (?, ?, ?)`); this.insertVec = db.prepare(`INSERT INTO ${VEC_TABLE} (rowid, collection_id, embedding) VALUES (?, ?, ?)`); } @@ -243,12 +256,24 @@ export class PartitionWriter { } } +/** + * Lookup of stored vector bytes by (hash, seq) from whichever partition holds + * the chunk, with its two statements prepared once for callers that loop. + */ +export function storedEmbeddingLookup(db: Database): (hash: string, seq: number) => Uint8Array | undefined { + const rowOf = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? LIMIT 1`); + const vectorOf = db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`); + return (hash, seq) => { + const row = rowOf.get(hash, seq) as { id: number } | undefined; + if (!row) return undefined; + const stored = vectorOf.get(vecInteger(row.id)) as { embedding: Uint8Array } | undefined; + return stored?.embedding; + }; +} + /** Stored vector bytes of (hash, seq) from whichever partition holds it. */ export function storedEmbedding(db: Database, hash: string, seq: number): Uint8Array | undefined { - const row = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ? LIMIT 1`).get(hash, seq) as { id: number } | undefined; - if (!row) return undefined; - const stored = db.prepare(`SELECT embedding FROM ${VEC_TABLE} WHERE rowid = ?`).get(vecInteger(row.id)) as { embedding: Uint8Array } | undefined; - return stored?.embedding; + return storedEmbeddingLookup(db)(hash, seq); } /** (hash, seq) behind a vec0 rowid. */ From 5c7367d1a39b346e96649cc8842415c081985f45 Mon Sep 17 00:00:00 2001 From: Brett Date: Fri, 4 Sep 2026 16:58:29 -0500 Subject: [PATCH 35/82] fix(review): apply the code-review findings on the partitioned vector layout Findings from the ce-code-review run 20260904-161233-cdf84496, all applied: - The migration's verification pass, the vectorless content_vectors delete, the cursor delete, the version stamp and the DROP of the legacy table now run in one IMMEDIATE transaction, so a writer on an older build cannot land a row in the legacy table between the delete and the drop and lose its only vector. The partitioned table is created with IF NOT EXISTS inside a double-checked transaction, so two first openers no longer race on CREATE VIRTUAL TABLE. `runStoreMigrations` also runs the copy when a legacy table exists on an already-stamped store, as the module's own comment claimed; when that repair fails on a dimension mismatch it warns and defers instead of blocking every open. - `qmd update`, `qmd collection add` and the library's `update()` now run the copy pass, so a document that appears in a second collection is searchable there right after indexing instead of after the next embed run, and the library's `update()` also removes stale partition rows and reports both counts (`staleVectorsRemoved`, `vectorsCopied`). - `removeCollection` runs as one IMMEDIATE transaction, so an interrupted removal cannot leave the vec0 rows, the mapping and the collection id out of step. - `copyVectorsToNewCollections` commits in slices of 1,000 rows, so a collection that gains many already-embedded documents no longer holds the write lock for the whole copy; a hash queued for re-embedding in one slice is skipped by later ones. Tests cover each change: the flip holds one write lock, a row written after the copy phase still lands in a partition, a second opener finishes a migration another opener started, a legacy table on a stamped store is migrated at open, `qmd update` and the SDK `update()` copy vectors into a collection that gained a hash, `removeCollection` rolls back as a unit, and a 2,500-row copy commits in batches with straddling hashes queued correctly. (cherry picked from commit 1921727ac82bf46943d13293d0427082e56b0e68) --- src/cli/qmd.ts | 12 ++ src/index.ts | 14 +++ src/store-migrations.ts | 93 +++++++++----- src/store.ts | 83 ++++++++----- src/vec-layout.ts | 2 +- test/cli.test.ts | 57 ++++++++- test/sdk.test.ts | 57 +++++++++ test/store-migrations.test.ts | 116 ++++++++++++++++++ test/store.test.ts | 223 +++++++++++++++++++++++++++++++--- 9 files changed, 576 insertions(+), 81 deletions(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index c9a7202df..ddeb20b3f 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -58,6 +58,7 @@ import { getActiveDocumentPaths, cleanupOrphanedContent, cleanupOrphanedVectors, + copyVectorsToNewCollections, countOrphanedVectors, previewCleanup, runCleanup, @@ -1021,6 +1022,10 @@ async function updateCollections(): Promise { // filter cannot see documents.active, so those rows would take k slots // from a scoped search until they are removed. const staleVectors = cleanupOrphanedVectors(db); + // The pending count below only sees content_vectors, so a hash that joined + // a collection while already embedded elsewhere would be neither counted + // nor searchable there; copying its rows closes that gap without a model. + const copiedVectors = copyVectorsToNewCollections(db).copied; // Check if any documents need embedding (show once at end) const needsEmbedding = getHashesNeedingEmbedding(db); @@ -1030,6 +1035,9 @@ async function updateCollections(): Promise { if (staleVectors > 0) { console.log(`Removed ${staleVectors} stale vector row(s)`); } + if (copiedVectors > 0) { + console.log(`Copied ${copiedVectors} vector(s) into collections that gained already-embedded documents`); + } if (needsEmbedding > 0) { console.log(`\nRun 'qmd embed' to update embeddings (${needsEmbedding} unique hashes need vectors)`); } @@ -2237,6 +2245,7 @@ async function indexFiles(pwd?: string, globPattern: string = DEFAULT_GLOB, coll // Clean up orphaned content hashes (content not referenced by any document) const orphanedContent = cleanupOrphanedContent(db); + const copiedVectors = copyVectorsToNewCollections(db, collectionName).copied; // Check if vector index needs updating const needsEmbedding = getHashesNeedingEmbedding(db); @@ -2248,6 +2257,9 @@ async function indexFiles(pwd?: string, globPattern: string = DEFAULT_GLOB, coll if (orphanedContent > 0) { console.log(`Cleaned up ${orphanedContent} orphaned content hash(es)`); } + if (copiedVectors > 0) { + console.log(`Copied ${copiedVectors} vector(s) into collections that gained already-embedded documents`); + } if (needsEmbedding > 0 && !suppressEmbedNotice) { console.log(`\nRun 'qmd embed' to update embeddings (${needsEmbedding} unique hashes need vectors)`); diff --git a/src/index.ts b/src/index.ts index 1e7fefcb0..e3b08afe9 100644 --- a/src/index.ts +++ b/src/index.ts @@ -41,6 +41,7 @@ import { vacuumDatabase, cleanupOrphanedContent, cleanupOrphanedVectors, + copyVectorsToNewCollections, deleteLLMCache, deleteInactiveDocuments, clearAllEmbeddings, @@ -206,6 +207,10 @@ export type UpdateResult = { unchanged: number; removed: number; skipped: number; + /** Vector rows removed because their (hash, collection) no longer has an active document. */ + staleVectorsRemoved: number; + /** Vector rows copied into the partition of a collection that gained an already-embedded hash. */ + vectorsCopied: number; needsEmbedding: number; }; @@ -622,6 +627,13 @@ export async function createStore(options: StoreOptions): Promise { totalSkipped += result.skipped; } + // A changed file rewrites its document's hash in place and strands the + // old hash's partition rows, which take k slots from scoped searches; + // a hash that joined a collection while embedded elsewhere stays + // unsearchable there until its rows are copied. + const staleVectorsRemoved = cleanupOrphanedVectors(db); + const vectorsCopied = copyVectorsToNewCollections(db).copied; + return { collections: filtered.length, indexed: totalIndexed, @@ -629,6 +641,8 @@ export async function createStore(options: StoreOptions): Promise { unchanged: totalUnchanged, removed: totalRemoved, skipped: totalSkipped, + staleVectorsRemoved, + vectorsCopied, needsEmbedding: internal.getHashesNeedingEmbedding(), }; }, diff --git a/src/store-migrations.ts b/src/store-migrations.ts index 172aac097..cb938ccf0 100644 --- a/src/store-migrations.ts +++ b/src/store-migrations.ts @@ -10,8 +10,9 @@ * per IMMEDIATE transaction, and keeps the last copied chunk_id in * store_config. A killed process resumes from that cursor, and two processes * that open the same database take turns on chunks because each reads the - * cursor inside its own write transaction. The legacy table is dropped in the - * same transaction that stamps the version, so that drop is the flip. + * cursor inside its own write transaction. A verification pass that copies + * rows the walk missed, the version stamp and the drop of the legacy table + * share one IMMEDIATE transaction, so that drop is the flip. */ import type { Database } from "./db.js"; @@ -152,18 +153,17 @@ function copyLegacyChunks( /** * Rows the chunk walk can miss: a pre-upgrade `qmd cleanup` repacking the * legacy table moves live rows into its newest chunk while the walk is past - * it. They are copied by key from the legacy table. + * it, and a process on the old layout keeps writing into chunks the walk has + * left behind. They are copied by key from the legacy table. Runs inside the + * caller's write transaction. */ function copyStragglers(db: Database, legacy: ReadableVecLayout): void { const legacyVector = db.prepare(`SELECT embedding FROM ${legacy.table} WHERE hash_seq = ?`); - db.transaction(() => { - if (vecLayout(db).kind !== "legacy") return; - const writer = new PartitionWriter(db); - for (const missing of missingPartitionRows(db)) { - const row = legacyVector.get(`${missing.hash}_${missing.seq}`) as { embedding: Uint8Array } | undefined; - if (row) writer.writeToCollection(missing.hash, missing.seq, missing.collection, row.embedding); - } - }).immediate(); + const writer = new PartitionWriter(db); + for (const missing of missingPartitionRows(db)) { + const row = legacyVector.get(`${missing.hash}_${missing.seq}`) as { embedding: Uint8Array } | undefined; + if (row) writer.writeToCollection(missing.hash, missing.seq, missing.collection, row.embedding); + } } /** @@ -177,28 +177,51 @@ export function deleteVectorlessContentVectors(db: Database): number { `).run().changes; } -function flipToPartitioned(db: Database): void { +/** + * The straggler copy, the vectorless cleanup and the drop hold one write + * lock: a legacy row written between them by a process on the old layout + * would otherwise leave a content_vectors row whose only vector is in the + * dropped table, which the pending detector never re-embeds. `onFlip` runs + * with that lock held, after the verification and before the drop. + */ +function flipToPartitioned(db: Database, legacy: ReadableVecLayout, onFlip: () => void): void { db.transaction(() => { if (vecLayout(db).kind !== "legacy") return; + copyStragglers(db, legacy); + deleteVectorlessContentVectors(db); + onFlip(); db.prepare(`DELETE FROM store_config WHERE key = ?`).run(CURSOR_KEY); db.exec(`PRAGMA user_version = ${Math.max(getUserVersion(db), VECTOR_PARTITION_VERSION)}`); db.exec(`DROP TABLE IF EXISTS ${LEGACY_VEC_TABLE}`); }).immediate(); } +/** Dimensions of the partitioned table; undefined when there is no table. */ +function partitionedTableDimensions(db: Database): number | null | undefined { + const row = db.prepare(`SELECT sql FROM sqlite_master WHERE type = 'table' AND name = ?`).get(VEC_TABLE) as { sql: string } | undefined; + return row ? parseDimensions(row.sql) : undefined; +} + +/** + * Check and create share one IMMEDIATE transaction with a double-checked + * read, matching applyVersionedStep: concurrent first openers create the + * partitioned table once and the losers reuse it. + */ function ensurePartitionedTable(db: Database, dimensions: number): void { - const layout = vecLayout(db); - const existing = db.prepare(`SELECT sql FROM sqlite_master WHERE type = 'table' AND name = ?`).get(VEC_TABLE) as { sql: string } | undefined; - if (!existing) { - createPartitionedVecTable(db, dimensions); - return; - } - const existingDims = parseDimensions(existing.sql); - if (existingDims !== dimensions) { - throw new Error( - `Vector migration found a ${VEC_TABLE} table with ${existingDims ?? "unknown"} dimensions while the legacy ${layout.kind} table holds ${dimensions}d vectors` - ); - } + const exists = (): boolean => { + const existing = partitionedTableDimensions(db); + if (existing === undefined) return false; + if (existing !== dimensions) { + throw new Error( + `Vector migration found a ${VEC_TABLE} table with ${existing ?? "unknown"} dimensions while the legacy ${LEGACY_VEC_TABLE} table holds ${dimensions}d vectors` + ); + } + return true; + }; + if (exists()) return; + db.transaction(() => { + if (!exists()) createPartitionedVecTable(db, dimensions); + }).immediate(); } /** @@ -230,10 +253,7 @@ export function migrateVectorLayout(db: Database, options: VectorMigrationOption copyLegacyChunks(db, layout, dimensions, (copied) => report("copy", copied)); if (vecLayout(db).kind === "legacy") { report("verify", total); - copyStragglers(db, layout); - deleteVectorlessContentVectors(db); - report("flip", total); - flipToPartitioned(db); + flipToPartitioned(db, layout, () => report("flip", total)); report("vacuum", total); try { db.exec(`VACUUM`); @@ -249,9 +269,26 @@ export type StoreMigrationDeps = VectorMigrationOptions & { installFtsSyncTriggers: (db: Database) => void; }; +/** + * The vector step also runs on a stamped database that holds a legacy table: + * an older build recreates `vectors_vec` on any database it embeds into, and + * the resolver reports that table as the layout until it is gone. + */ export function runStoreMigrations(db: Database, deps: StoreMigrationDeps): void { applyVersionedStep(db, FTS_SYNC_TRIGGERS_VERSION, () => deps.installFtsSyncTriggers(db)); if (getUserVersion(db) < VECTOR_PARTITION_VERSION) { migrateVectorLayout(db, deps); + return; + } + // A legacy table on a stamped store came from an older build; copying it + // here is a repair, so a failure (a dimension mismatch with the partitioned + // table) degrades vector search instead of blocking every open. The embed + // path raises the actionable error when it next runs. + if (vecLayout(db).kind === "legacy") { + try { + migrateVectorLayout(db, deps); + } catch (err) { + console.warn(`Legacy vector table left in place: ${err instanceof Error ? err.message : String(err)}`); + } } } diff --git a/src/store.ts b/src/store.ts index 2203b2a18..cb67c2f2f 100644 --- a/src/store.ts +++ b/src/store.ts @@ -36,6 +36,7 @@ import { vecInteger, vecLayout, vecTableReadable, + type MissingPartitionRow, } from "./vec-layout.js"; import { FTS_SYNC_TRIGGERS_VERSION, @@ -4089,29 +4090,31 @@ function deleteVectorPartition(db: Database, collectionName: string): void { } /** - * Remove a collection and clean up its documents. - * Uses collections.ts to remove from YAML config and cleans up database. + * Remove a collection: its vector partition, its documents, the content no + * active document references any more, and its store_collections row. */ export function removeCollection(db: Database, collectionName: string): { deletedDocs: number; cleanedHashes: number } { - deleteVectorPartition(db, collectionName); + // One commit: a partition delete that lands without the documents delete + // leaves the mapping and id row out of step with the vec0 table, and the + // update-time orphan cleanup skips rows whose documents are still active. + return db.transaction(() => { + deleteVectorPartition(db, collectionName); - // Delete documents from database - const docResult = db.prepare(`DELETE FROM documents WHERE collection = ?`).run(collectionName); - db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(collectionName); + const docResult = db.prepare(`DELETE FROM documents WHERE collection = ?`).run(collectionName); + db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(collectionName); - // Clean up orphaned content hashes - const cleanupResult = db.prepare(` - DELETE FROM content - WHERE hash NOT IN (SELECT DISTINCT hash FROM documents WHERE active = 1) - `).run(); + const cleanupResult = db.prepare(` + DELETE FROM content + WHERE hash NOT IN (SELECT DISTINCT hash FROM documents WHERE active = 1) + `).run(); - // Remove from store_collections - deleteStoreCollection(db, collectionName); + deleteStoreCollection(db, collectionName); - return { - deletedDocs: docResult.changes, - cleanedHashes: cleanupResult.changes - }; + return { + deletedDocs: docResult.changes, + cleanedHashes: cleanupResult.changes, + }; + }).immediate(); } /** @@ -4884,39 +4887,57 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { }).immediate()); } +/** Partition rows written per commit by copyVectorsToNewCollections. */ +export const VECTOR_COPY_BATCH_ROWS = 1000; + /** * Give every active (hash, collection) pair whose chunks are embedded a row * in that collection's partition, copied from any partition that holds the * chunk, so a hash that gains a collection is searchable there without * another model call. A chunk no partition holds is queued for embedding by * deleting its hash's rows, which the pending detector then picks up. + * + * Rows commit in batches of VECTOR_COPY_BATCH_ROWS so a collection that gains + * thousands of embedded hashes does not hold the write lock for the whole + * copy; an interrupted run resumes from whatever missingPartitionRows still + * reports. */ export function copyVectorsToNewCollections(db: Database, collection?: string): { copied: number; queued: number } { const layout = vecLayout(db); if (layout.kind === "legacy" || !isSqliteVecAvailable()) return { copied: 0, queued: 0 }; - return withLazyContentVectorMigration(db, () => db.transaction(() => { + return withLazyContentVectorMigration(db, () => { const missing = missingPartitionRows(db, collection); if (missing.length === 0) return { copied: 0, queued: 0 }; const writer = layout.kind === "partitioned" ? new PartitionWriter(db) : null; const storedVector = writer ? storedEmbeddingLookup(db) : null; + const deleteChunks = db.prepare(`DELETE FROM content_vectors WHERE hash = ?`); const queued = new Set(); let copied = 0; - for (const row of missing) { - if (queued.has(row.hash)) continue; - const vector = storedVector?.(row.hash, row.seq); - if (!vector || !writer) { - queued.add(row.hash); - continue; + const copyBatch = (rows: MissingPartitionRow[]) => { + const queuedNow: string[] = []; + for (const row of rows) { + if (queued.has(row.hash)) continue; + const vector = storedVector?.(row.hash, row.seq); + if (!vector || !writer) { + queued.add(row.hash); + queuedNow.push(row.hash); + continue; + } + if (writer.writeToCollection(row.hash, row.seq, row.collection, vector)) copied++; } - if (writer.writeToCollection(row.hash, row.seq, row.collection, vector)) copied++; - } - const deleteChunks = db.prepare(`DELETE FROM content_vectors WHERE hash = ?`); - for (const hash of queued) { - deletePartitionRowsOfHash(db, hash); - deleteChunks.run(hash); + // Deleting by hash also removes rows an earlier batch already committed + // for it; the run-level set keeps a later batch from copying it again. + for (const hash of queuedNow) { + deletePartitionRowsOfHash(db, hash); + deleteChunks.run(hash); + } + }; + for (let start = 0; start < missing.length; start += VECTOR_COPY_BATCH_ROWS) { + const batch = missing.slice(start, start + VECTOR_COPY_BATCH_ROWS); + db.transaction(() => copyBatch(batch)).immediate(); } return { copied, queued: queued.size }; - }).immediate()); + }); } /** diff --git a/src/vec-layout.ts b/src/vec-layout.ts index e28ec16c1..3242a58b6 100644 --- a/src/vec-layout.ts +++ b/src/vec-layout.ts @@ -120,7 +120,7 @@ export function createVectorMetadataTables(db: Database): void { export function createPartitionedVecTable(db: Database, dimensions: number): void { db.exec( - `CREATE VIRTUAL TABLE ${VEC_TABLE} USING vec0(collection_id INTEGER PARTITION KEY, embedding float[${dimensions}] distance_metric=cosine)` + `CREATE VIRTUAL TABLE IF NOT EXISTS ${VEC_TABLE} USING vec0(collection_id INTEGER PARTITION KEY, embedding float[${dimensions}] distance_metric=cosine)` ); } diff --git a/test/cli.test.ts b/test/cli.test.ts index 18959f9b5..bb23030d6 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -16,7 +16,7 @@ import { setTimeout as sleep } from "timers/promises"; import { buildEditorUri, termLink, resolveEmbedModelForCli } from "../src/cli/qmd.ts"; import { openDatabase } from "../src/db.ts"; import { createStore, insertContent, insertDocument } from "../src/store.ts"; -import { VEC_ROWS_TABLE, VEC_TABLE } from "../src/vec-layout.ts"; +import { VEC_COLLECTION_IDS_TABLE, VEC_ROWS_TABLE, VEC_TABLE } from "../src/vec-layout.ts"; import { DEFAULT_EMBED_MODEL_URI, DEFAULT_GENERATE_MODEL_URI, DEFAULT_RERANK_MODEL_URI } from "../src/llm.ts"; import { setConfigSource } from "../src/collections.ts"; @@ -1402,6 +1402,51 @@ describe("qmd update stale vector rows", () => { db.close(); } }); + + test("a collection that gains an already-embedded document receives its vector rows at the end of update", async () => { + const env = await createIsolatedTestEnv("copied-vector-rows"); + const firstDir = join(testDir, `copied-vector-rows-first-${testCounter}`); + const secondDir = join(testDir, `copied-vector-rows-second-${testCounter}`); + await mkdir(firstDir, { recursive: true }); + await mkdir(secondDir, { recursive: true }); + await writeFile(join(firstDir, "shared.md"), "# Shared\n\nembedded once, then joins a second collection\n"); + await writeFile(join(secondDir, "other.md"), "# Other\n\nonly in the second collection\n"); + + const addFirst = await runQmd(["collection", "add", firstDir, "--name", "first"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(addFirst.exitCode).toBe(0); + const addSecond = await runQmd(["collection", "add", secondDir, "--name", "second"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(addSecond.exitCode).toBe(0); + + const store = createStore(env.dbPath); + let sharedHash: string; + try { + const now = new Date().toISOString(); + sharedHash = (store.db.prepare(`SELECT hash FROM documents WHERE collection = 'first' AND path = 'shared.md' AND active = 1`).get() as { hash: string }).hash; + store.ensureVecTable(3); + store.insertEmbedding(sharedHash, 0, 0, new Float32Array([1, 2, 3]), "test", now); + } finally { + store.close(); + } + + await copyFile(join(firstDir, "shared.md"), join(secondDir, "shared.md")); + + const update = await runQmd(["update"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(update.exitCode).toBe(0); + expect(update.stdout).toContain("Copied 1 vector(s) into collections that gained already-embedded documents"); + + const db = openDatabase(env.dbPath); + try { + const partitions = db.prepare(` + SELECT ci.name AS collection FROM ${VEC_ROWS_TABLE} vr + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + WHERE vr.hash = ? + ORDER BY ci.name + `).all(sharedHash); + expect(partitions).toEqual([{ collection: "first" }, { collection: "second" }]); + } finally { + db.close(); + } + }); }); describe("orphaned embedding vectors (#768)", () => { @@ -1471,6 +1516,14 @@ describe("orphaned embedding vectors (#768)", () => { expect(stdout).not.toContain("stale vector row"); const status = await runQmd(["status"], { dbPath: localDbPath, configDir: localConfigDir }); expect(status.stdout).toMatch(/Orphaned:\s+3 embedding chunks/); + + const db = openDatabase(localDbPath); + const vectorlessLiveChunks = (db.prepare(` + SELECT COUNT(*) as c FROM content_vectors cv + WHERE EXISTS (SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1) + `).get() as { c: number }).c; + db.close(); + expect(vectorlessLiveChunks).toBe(0); }); test("cleanup --dry-run reports what would be removed without deleting", async () => { @@ -1491,7 +1544,7 @@ describe("orphaned embedding vectors (#768)", () => { const cache = (db.prepare(`SELECT COUNT(*) as c FROM llm_cache`).get() as { c: number }).c; const inactive = (db.prepare(`SELECT COUNT(*) as c FROM documents WHERE active = 0`).get() as { c: number }).c; db.close(); - expect(vectors).toBe(4); + expect(vectors).toBe(3); expect(cache).toBe(2); expect(inactive).toBe(1); }); diff --git a/test/sdk.test.ts b/test/sdk.test.ts index 5ae0d0e01..84bff242d 100644 --- a/test/sdk.test.ts +++ b/test/sdk.test.ts @@ -23,6 +23,7 @@ import { type ExpandQueryOptions, } from "../src/index.js"; import { setDefaultLlamaCpp } from "../src/llm.js"; +import { VEC_COLLECTION_IDS_TABLE, VEC_ROWS_TABLE } from "../src/vec-layout.js"; // ============================================================================= // Test Helpers @@ -850,6 +851,8 @@ describe("update", () => { expect(result.unchanged).toBe(0); expect(result.removed).toBe(0); expect(result.skipped).toBe(0); + expect(result.staleVectorsRemoved).toBe(0); + expect(result.vectorsCopied).toBe(0); expect(typeof result.needsEmbedding).toBe("number"); await store.close(); @@ -1101,6 +1104,60 @@ describe("embed", () => { } }); + test("store.update drops stale vector rows and copies rows into a collection that gained an embedded hash", async () => { + const leftDir = join(testDir, `vector-rows-left-${Date.now()}`); + const rightDir = join(testDir, `vector-rows-right-${Date.now()}`); + await mkdir(leftDir, { recursive: true }); + await mkdir(rightDir, { recursive: true }); + const shared = "# Shared\n\nEmbedded once, then joins a second collection.\n"; + await writeFile(join(leftDir, "shared.md"), shared); + await writeFile(join(rightDir, "gone.md"), "# Gone\n\nDisappears before the second update.\n"); + + const store = await createStore({ + dbPath: freshDbPath(), + config: { + collections: { + left: { path: leftDir, pattern: "**/*.md" }, + right: { path: rightDir, pattern: "**/*.md" }, + }, + }, + }); + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.internal.llm = createFakeEmbedLlm() as any; + + const partitionsOf = (path: string): string[] => { + const doc = store.internal.db.prepare(`SELECT hash FROM documents WHERE path = ? LIMIT 1`).get(path) as { hash: string }; + const rows = store.internal.db.prepare(` + SELECT ci.name AS collection FROM ${VEC_ROWS_TABLE} vr + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + WHERE vr.hash = ? + ORDER BY ci.name + `).all(doc.hash) as Array<{ collection: string }>; + return rows.map((row) => row.collection); + }; + + try { + await store.update(); + await store.embed(); + expect(partitionsOf("shared.md")).toEqual(["left"]); + expect(partitionsOf("gone.md")).toEqual(["right"]); + + await writeFile(join(rightDir, "shared.md"), shared); + await rm(join(rightDir, "gone.md")); + const result = await store.update(); + + expect(result.removed).toBe(1); + expect(result.staleVectorsRemoved).toBe(1); + expect(result.vectorsCopied).toBe(1); + expect(result.needsEmbedding).toBe(0); + expect(partitionsOf("shared.md")).toEqual(["left", "right"]); + expect(partitionsOf("gone.md")).toEqual([]); + } finally { + setDefaultLlamaCpp(null); + await store.close(); + } + }); + test("store.embed rejects invalid batch limits", async () => { const store = await createStore({ dbPath: freshDbPath(), diff --git a/test/store-migrations.test.ts b/test/store-migrations.test.ts index 9aa6956d4..466e6b69e 100644 --- a/test/store-migrations.test.ts +++ b/test/store-migrations.test.ts @@ -13,12 +13,15 @@ import { VECTOR_PARTITION_VERSION, getUserVersion, migrateVectorLayout, + runStoreMigrations, + type VectorMigrationPhase, type VectorMigrationProgress, } from "../src/store-migrations.js"; import { LEGACY_VEC_TABLE, VEC_ROWS_TABLE, VEC_TABLE, + createPartitionedVecTable, resolveCollectionId, vecInteger, vecLayout, @@ -312,6 +315,91 @@ describe("migrateVectorLayout", () => { expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS + 1); }); + test("a legacy row written after the copy phase and before the flip ends up in the partition", async () => { + const s = await openStore(); + const fixture = seedStandardFixture(s.db); + const phases: VectorMigrationPhase[] = []; + let injected = false; + + expect(migrateVectorLayout(s.db, { + sqliteVecAvailable: true, + onProgress: (p) => { + phases.push(p.phase); + if (p.phase === "verify" && !injected) { + injected = true; + fixture.embedded("b", "late", [vec(0.7, 0.7, 0.1)]); + } + }, + })).toBe("applied"); + + expect(injected).toBe(true); + expect(phases.slice(phases.indexOf("verify"))).toEqual(["verify", "flip", "vacuum", "done"]); + expect(storedVector(s.db, "late", 0, "b")).toEqual(Array.from(vec(0.7, 0.7, 0.1))); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS + 1); + expect(s.db.prepare(`SELECT 1 FROM content_vectors WHERE hash = 'late' AND seq = 0`).get()).toBeTruthy(); + expect(tableNames(s.db, LEGACY_VEC_TABLE)).toEqual([]); + }); + + test("the verification pass and the drop hold one write lock", async () => { + const s = await openStore(); + seedStandardFixture(s.db); + const other = openSecondConnection(s); + other.exec(`PRAGMA busy_timeout = 0`); + let lockedAtFlip: boolean | null = null; + + expect(migrateVectorLayout(s.db, { + sqliteVecAvailable: true, + onProgress: (p) => { + if (p.phase !== "flip") return; + try { + other.exec(`BEGIN IMMEDIATE`); + other.exec(`ROLLBACK`); + lockedAtFlip = false; + } catch { + lockedAtFlip = true; + } + }, + })).toBe("applied"); + + expect(lockedAtFlip).toBe(true); + expect(vecLayout(s.db).kind).toBe("partitioned"); + expect(tableNames(s.db, LEGACY_VEC_TABLE)).toEqual([]); + }); + + test("creating the partitioned table a second time is a no-op", async () => { + const s = await openStore(); + createPartitionedVecTable(s.db, DIMS); + expect(() => createPartitionedVecTable(s.db, DIMS)).not.toThrow(); + expect(vecLayout(s.db)).toMatchObject({ kind: "partitioned", dimensions: DIMS }); + expect(tableNames(s.db, VEC_TABLE)).toContain(`${VEC_TABLE}_chunks`); + }); + + test("a second opener finishes the migration with the partitioned table another opener created", async () => { + const s = await openStore(); + seedStandardFixture(s.db); + createPartitionedVecTable(s.db, DIMS); + expect(vecLayout(s.db).kind).toBe("legacy"); + const second = openSecondConnection(s); + + expect(migrateVectorLayout(second, { sqliteVecAvailable: true })).toBe("applied"); + + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db)).toMatchObject({ kind: "partitioned", dimensions: DIMS }); + expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS); + expect(migrateVectorLayout(s.db, { sqliteVecAvailable: true })).toBe("applied"); + }); + + test("refuses a partitioned table whose dimensions differ from the legacy table's", async () => { + const s = await openStore(); + seedStandardFixture(s.db); + createPartitionedVecTable(s.db, DIMS + 1); + + expect(() => migrateVectorLayout(s.db, { sqliteVecAvailable: true })).toThrow(/dimensions/); + expect(getUserVersion(s.db)).toBe(1); + expect(vecLayout(s.db).kind).toBe("legacy"); + }); + test("the flip drops the legacy table and its shadow tables with the version stamp", async () => { const s = await openStore(); seedStandardFixture(s.db); @@ -326,3 +414,31 @@ describe("migrateVectorLayout", () => { expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); }); }); + +describe("runStoreMigrations", () => { + test("migrates a legacy table an older build created on a stamped store", async () => { + const s = await openStore(); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + seedStandardFixture(s.db); + s.db.exec(`PRAGMA user_version = ${VECTOR_PARTITION_VERSION}`); + expect(vecLayout(s.db).kind).toBe("legacy"); + + runStoreMigrations(s.db, { installFtsSyncTriggers: () => {}, sqliteVecAvailable: true }); + + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db)).toMatchObject({ kind: "partitioned", dimensions: DIMS }); + expect(tableNames(s.db, LEGACY_VEC_TABLE)).toEqual([]); + expect(partitionCount(s.db, "a")).toBe(STANDARD_A_ROWS); + expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS); + }); + + test("leaves a stamped store without a legacy table alone", async () => { + const s = await openStore(); + createPartitionedVecTable(s.db, DIMS); + + runStoreMigrations(s.db, { installFtsSyncTriggers: () => { throw new Error("must not reinstall"); }, sqliteVecAvailable: true }); + + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db)).toMatchObject({ kind: "partitioned", dimensions: DIMS }); + }); +}); diff --git a/test/store.test.ts b/test/store.test.ts index 2e7ed22ce..a19634daf 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -19,6 +19,7 @@ import * as llmModule from "../src/llm.js"; import { disposeDefaultLlamaCpp, setDefaultLlamaCpp } from "../src/llm.js"; import { createStore, + DEFAULT_EMBED_MODEL, DEFAULT_QUERY_MODEL, DEFAULT_RERANK_MODEL, verifySqliteVecLoaded, @@ -59,6 +60,8 @@ import { insertDocument, cleanupOrphanedVectors, clearAllEmbeddings, + copyVectorsToNewCollections, + VECTOR_COPY_BATCH_ROWS, maybeAdoptLegacyEmbeddingFingerprint, removeCollection, renameCollection, @@ -75,7 +78,7 @@ import { type RankedListMeta, } from "../src/store.js"; import type { CollectionConfig } from "../src/collections.js"; -import { VEC_ROWS_TABLE, VEC_TABLE, deletePartitionRows, resolveCollectionId } from "../src/vec-layout.js"; +import { VEC_ROWS_TABLE, VEC_TABLE, deletePartitionRows, resolveCollectionId, vecInteger } from "../src/vec-layout.js"; // ============================================================================= // LlamaCpp Setup @@ -191,6 +194,35 @@ async function insertTestDocument( return row?.id ?? 0; } +/** + * Same connection, but any statement whose SQL contains `fragment` throws + * `message` instead of running, whether prepared or exec'd. + */ +function failingDb(db: Database, fragment: string, message: string): Database { + const guard = (sql: string) => { + if (sql.includes(fragment)) throw new Error(message); + }; + return { + prepare: (sql: string) => { + guard(sql); + return db.prepare(sql); + }, + transaction: (fn) => db.transaction(fn), + exec: (sql: string) => { + guard(sql); + db.exec(sql); + }, + loadExtension: (path: string) => db.loadExtension(path), + close: () => db.close(), + }; +} + +function vectorRowCount(store: Store, collection: string): number { + const id = resolveCollectionId(store.db, collection); + if (id === undefined) return 0; + return (store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE} WHERE collection_id = ?`).get(id) as { c: number }).c; +} + /** Sync YAML config file to SQLite store_collections in the current test store */ async function syncTestConfig(): Promise { if (!currentTestStore) return; @@ -3877,18 +3909,7 @@ describe("cleanupOrphanedVectors atomicity", () => { // Fault injection: same connection, but the content_vectors DELETE throws — // after the vector DELETEs already executed inside the transaction. function makeFailingDb(db: Database): Database { - return { - prepare: (sql: string) => db.prepare(sql), - transaction: (fn) => db.transaction(fn), - exec: (sql: string) => { - if (sql.includes("DELETE FROM content_vectors")) { - throw new Error("injected failure between deletes"); - } - return db.exec(sql); - }, - loadExtension: (path: string) => db.loadExtension(path), - close: () => db.close(), - }; + return failingDb(db, "DELETE FROM content_vectors", "injected failure between deletes"); } test("removes orphaned chunks from both tables and returns the count", async () => { @@ -4530,12 +4551,6 @@ describe("Vector Search collection filter", () => { } } - function vectorRowCount(store: Store, collection: string): number { - const id = resolveCollectionId(store.db, collection); - if (id === undefined) return 0; - return (store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE} WHERE collection_id = ?`).get(id) as { c: number }).c; - } - test("searchVec scoped to two of three collections returns exactly those two", async () => { const store = await createTestStore(); const first = await createTestCollection({ name: "first", pwd: "/test/first" }); @@ -4876,6 +4891,40 @@ describe("Vector Search collection filter", () => { await cleanupTestDb(store); }); + test("removeCollection rolls back the partition delete when the documents delete fails", async () => { + const store = await createTestStore(); + const gone = await createTestCollection({ name: "gone", pwd: "/test/gone" }); + const kept = await createTestCollection({ name: "kept", pwd: "/test/kept" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, gone, "gonehash", [vector(1, 0), vector(0.9, 0.1)]); + await insertVecDoc(store, kept, "kepthash", [vector(0.6, 0.8)]); + const goneId = resolveCollectionId(store.db, gone)!; + const partitionVectors = (id: number) => + (store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE} WHERE collection_id = ?`).get(vecInteger(id)) as { c: number }).c; + const documentsIn = (collection: string) => + (store.db.prepare(`SELECT COUNT(*) AS c FROM documents WHERE collection = ?`).get(collection) as { c: number }).c; + + const failing = failingDb(store.db, "DELETE FROM documents WHERE collection = ?", "injected failure after the partition delete"); + expect(() => removeCollection(failing, gone)).toThrow("injected failure after the partition delete"); + + // The partition delete ran first in the same transaction; none of it survives the rollback. + expect(resolveCollectionId(store.db, gone)).toBe(goneId); + expect(vectorRowCount(store, gone)).toBe(2); + expect(partitionVectors(goneId)).toBe(2); + expect(documentsIn(gone)).toBe(1); + expect((await store.searchVec("ignored", "test-model", 5, gone, undefined, query)).map((r) => r.hash)).toEqual(["gonehash"]); + + // The connection is left clean: a plain retry removes everything. + removeCollection(store.db, gone); + expect(resolveCollectionId(store.db, gone)).toBeUndefined(); + expect(vectorRowCount(store, gone)).toBe(0); + expect(partitionVectors(goneId)).toBe(0); + expect(documentsIn(gone)).toBe(0); + expect(vectorRowCount(store, kept)).toBe(1); + + await cleanupTestDb(store); + }); + test("clearAllEmbeddings for one collection keeps a shared hash in every collection", async () => { const store = await createTestStore(); const cleared = await createTestCollection({ name: "cleared", pwd: "/test/cleared" }); @@ -5999,6 +6048,142 @@ describe("Embedding batching", () => { } }); + /** Active single-chunk documents for `hashes` in `collection`, in one transaction. */ + function seedDocs(db: Database, collection: string, hashes: readonly string[]): void { + const now = new Date().toISOString(); + db.transaction(() => { + for (const hash of hashes) { + insertContent(db, hash, `# ${hash}\n\nBody of ${hash}`, now); + insertDocument(db, collection, `${hash}.md`, hash, hash, now, now); + } + })(); + } + + /** `chunks` embedded chunks per hash under the default model, in every collection active for the hash, in one transaction. */ + function seedEmbeddings(store: Store, hashes: readonly string[], chunks = 1): void { + const now = new Date().toISOString(); + store.db.transaction(() => { + for (const hash of hashes) { + for (let seq = 0; seq < chunks; seq++) { + store.insertEmbedding(hash, seq, seq * 100, new Float32Array([1, 2, 3]), DEFAULT_EMBED_MODEL, now, chunks); + } + } + })(); + } + + function dropPartitionRows(db: Database, hash: string, seq: number): void { + const rows = db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ? AND seq = ?`).all(hash, seq) as { id: number }[]; + deletePartitionRows(db, rows.map((row) => row.id)); + } + + function rowsOfHash(db: Database, table: string, hash: string): number { + return (db.prepare(`SELECT COUNT(*) AS c FROM ${table} WHERE hash = ?`).get(hash) as { c: number }).c; + } + + /** Same connection, counting the IMMEDIATE transactions that run through it. */ + function commitCountingDb(db: Database, commits: { count: number }): Database { + return { + prepare: (sql: string) => db.prepare(sql), + exec: (sql: string) => db.exec(sql), + transaction: (fn) => { + const wrapped = db.transaction(fn); + const immediate = ((...args: Parameters) => { + const result = wrapped.immediate(...args); + commits.count++; + return result; + }) as typeof fn; + return Object.assign(((...args: Parameters) => wrapped(...args)) as typeof fn, { immediate }); + }, + loadExtension: (path: string) => db.loadExtension(path), + close: () => db.close(), + }; + } + + const paddedHash = (i: number) => `copy${String(i).padStart(4, "0")}`; + + test("generateEmbeddings copies more than one batch of vectors to a collection that gained embedded hashes", async () => { + const store = await createTestStore(); + const db = store.db; + const fakeLlm = createFakeEmbedLlm(); + + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.llm = fakeLlm as any; + + try { + const first = await createTestCollection({ name: "first", pwd: "/test/first" }); + const second = await createTestCollection({ name: "second", pwd: "/test/second" }); + const hashes = Array.from({ length: VECTOR_COPY_BATCH_ROWS * 2.5 }, (_, i) => paddedHash(i)); + store.ensureVecTable(3); + seedDocs(db, first, hashes); + seedEmbeddings(store, hashes); + seedDocs(db, second, hashes); + expect(store.getHashesNeedingEmbedding()).toBe(0); + + const result = await generateEmbeddings(store); + + expect(fakeLlm.embedBatchCalls).toHaveLength(0); + expect(result.chunksCopied).toBe(hashes.length); + expect(result.chunksEmbedded).toBe(0); + expect(vectorRowCount(store, second)).toBe(hashes.length); + const found = await store.searchVec("ignored", "test-model", 5, second, undefined, [1, 2, 3]); + expect(found).toHaveLength(5); + expect(found.every((r) => r.collectionName === second)).toBe(true); + expect(copyVectorsToNewCollections(db)).toEqual({ copied: 0, queued: 0 }); + } finally { + setDefaultLlamaCpp(null); + await cleanupTestDb(store); + } + }); + + test("copyVectorsToNewCollections queues a hash whose chunks straddle a batch boundary and keeps none of its copies", async () => { + const store = await createTestStore(); + const db = store.db; + try { + const first = await createTestCollection({ name: "first", pwd: "/test/first" }); + const second = await createTestCollection({ name: "second", pwd: "/test/second" }); + store.ensureVecTable(3); + + // missingPartitionRows orders by hash, seq, collection, so single-chunk + // hashes place each two-chunk hash's rows on both sides of a batch + // boundary. Suffix "a" sorts a hash right after the single it follows. + const batch = VECTOR_COPY_BATCH_ROWS; + const group1 = Array.from({ length: batch - 1 }, (_, i) => paddedHash(i)); + // Rows: (seq 0, second) closes batch 1; (seq 1, first) and (seq 1, second) open batch 2. + const copiedThenQueued = `${paddedHash(batch - 2)}a`; + const group2 = Array.from({ length: batch - 4 }, (_, i) => paddedHash(batch - 1 + i)); + // Rows: (seq 0, first) and (seq 0, second) close batch 2; (seq 1, second) opens batch 3. + const queuedThenSkipped = `${paddedHash(2 * batch - 6)}a`; + const singles = [...group1, ...group2]; + const straddlers = [copiedThenQueued, queuedThenSkipped]; + + seedDocs(db, first, [...singles, ...straddlers]); + seedEmbeddings(store, singles); + seedEmbeddings(store, straddlers, 2); + dropPartitionRows(db, copiedThenQueued, 1); + dropPartitionRows(db, queuedThenSkipped, 0); + seedDocs(db, second, [...singles, ...straddlers]); + const missingRows = singles.length + 3 + 3; + const commits = { count: 0 }; + + const result = copyVectorsToNewCollections(commitCountingDb(db, commits)); + + expect(commits.count).toBe(Math.ceil(missingRows / batch)); + expect(result.queued).toBe(2); + // Every single, plus copiedThenQueued's seq 0 written by batch 1 before batch 2 found its seq 1 missing. + expect(result.copied).toBe(singles.length + 1); + for (const hash of straddlers) { + expect(rowsOfHash(db, VEC_ROWS_TABLE, hash)).toBe(0); + expect(rowsOfHash(db, "content_vectors", hash)).toBe(0); + } + expect(vectorRowCount(store, first)).toBe(singles.length); + expect(vectorRowCount(store, second)).toBe(singles.length); + expect(store.getHashesNeedingEmbedding()).toBe(2); + expect(copyVectorsToNewCollections(db)).toEqual({ copied: 0, queued: 0 }); + } finally { + await cleanupTestDb(store); + } + }); + test("generateEmbeddings writes a batch in one transaction and rolls it back when a chunk write fails", async () => { const store = await createTestStore(); const real = store.db; From 84b953ed76ddfc0486e3668d891f03ef45fb6eb5 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 24 Sep 2026 19:44:37 -0500 Subject: [PATCH 36/82] fix(store): widen a vector search until it fills its limit A vector search asked each partition for `limit * 3` chunks and collapsed them to one row per file. When one long document held the nearest chunks, those slots all went to it and the search returned fewer documents than it asked for: one 20-chunk document and one single-chunk document gave 1 of 2 results at limit 2 and at limit 5, scoped or unscoped. The same happened when stale rows of a changed file sat nearer the query than any live document. The case was reported with a synthetic reproducer in a review comment on #936. Each scan target now resolves its own matches to documents. While they collapse into fewer than `limit` documents and the target holds rows beyond them, the KNN runs again with k doubled, up to sqlite-vec's 4096 cap. Each target returns its own nearest `limit` documents (or all it holds), so merging the targets by distance gives the exact nearest `limit` across the scope. The first KNN keeps `limit * 3`: on the live index, stars' top 60 chunks covered 46 to 60 distinct documents across ten sampled queries, so the second round is a guard for skewed data rather than a cost on typical queries. Step two is prepared once per search and leaves document bodies out: one long document can hold every match of a widened KNN, so only the final results load their body. The metadata predicate (current, error-free extraction plus the compiled filter) is built in one place for the eligible-row query and the document join. The store suite gains the one-long-document case, scoped and unscoped, and the metadata suite gains the reporter's two filtered cases: a long document with an ineligible copy of its content starving a second eligible document, and the best chunk of a 20- and a 450-chunk document surviving deduplication. All four returned one document before this change. The stale-row test's pre-cleanup search now answers instead of coming back empty. (cherry picked from commit d8ae76e6d1af3a554813377dc46e178db0441d2e) --- CHANGELOG.md | 9 +- src/store.ts | 189 +++++++++++++++++++++-------------- test/metadata-search.test.ts | 48 ++++++--- test/store.test.ts | 25 ++++- 4 files changed, 178 insertions(+), 93 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 82cd71288..297a7214b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -99,9 +99,12 @@ space of the old table. `qmd collection rename` no longer touches vectors, `qmd update` removes the vector rows of changed files as it finishes, and `qmd embed` copies an already-embedded document into a collection that - gains it instead of embedding it again. Downgrading to an older qmd - afterwards needs `qmd embed -f`, and a `qmd mcp` server started before the - upgrade must be restarted. + gains it instead of embedding it again. A metadata-filtered vector search + returns the nearest documents the filter admits however many chunks it + admits, and a vector search fills its limit even when one long document + holds all of the nearest chunks. Downgrading to an older qmd afterwards + needs `qmd embed -f`, and a `qmd mcp` server started before the upgrade + must be restarted. ## [2.8.3] - 2026-08-16 diff --git a/src/store.ts b/src/store.ts index cb67c2f2f..d66584e5d 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4619,14 +4619,38 @@ interface VecScanTarget { eligibleRowids?: readonly number[]; } +/** The document behind a vector match, at its nearest chunk. */ +interface VecDocumentMatch { + rowid: number; + hash: string; + pos: number; + filepath: string; + display_path: string; + title: string; + metadata_json: string | null; + distance: number; +} + +/** + * A metadata filter as SQL over a `document_metadata dm` join, admitting only + * documents whose metadata extraction is current and error-free. + */ +function compileCurrentMetadataFilter(filter: MetadataFilter): { sql: string; params: SQLiteValue[] } { + const compiled = compileMetadataFilter(filter, "d"); + return { + sql: `dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiled.sql}`, + params: compiled.params, + }; +} + /** * Prepares an exact top-k cosine scan of the vector table, run once per - * target. vec0 evaluates the partition equality inside the scan, so a small - * collection costs its own rows rather than the whole index and is never - * crowded out by a larger one (#775, #791, #803). A `rowid IN` restriction is - * applied inside the scan the same way, so a selective metadata filter gets an - * exact top-k of its own rows instead of whatever survives a post-filter of a - * larger top-k. + * target with k in 1..SQLITE_VEC_MAX_K. vec0 evaluates the partition equality + * inside the scan, so a small collection costs its own rows rather than the + * whole index and is never crowded out by a larger one (#775, #791, #803). A + * `rowid IN` restriction is applied inside the scan the same way, so a + * selective metadata filter gets an exact top-k of its own rows instead of + * whatever survives a post-filter of a larger top-k. */ function knnVecScanner(db: Database, partitioned: boolean, restricted: boolean): (embedding: Float32Array, k: number, target: VecScanTarget) => VecMatch[] { const conditions = ["embedding MATCH ?", "k = ?"]; @@ -4638,7 +4662,7 @@ function knnVecScanner(db: Database, partitioned: boolean, restricted: boolean): WHERE ${conditions.join(" AND ")} `); return (embedding, k, target) => { - const params: SQLiteValue[] = [embedding, Math.max(1, Math.min(SQLITE_VEC_MAX_K, k))]; + const params: SQLiteValue[] = [embedding, k]; if (partitioned) params.push(vecInteger(target.collectionId ?? 0)); if (restricted) params.push(rowidList(target.eligibleRowids ?? [])); return statement.all(...params) as VecMatch[]; @@ -4651,8 +4675,8 @@ function knnVecScanner(db: Database, partitioned: boolean, restricted: boolean): * holds the row's content passes the filter. */ function metadataEligibleVectorRows(db: Database, filter: MetadataFilter, collectionIds?: readonly number[]): Map { - const compiled = compileMetadataFilter(filter, "d"); - const params: SQLiteValue[] = [...compiled.params]; + const current = compileCurrentMetadataFilter(filter); + const params: SQLiteValue[] = [...current.params]; let scope = ""; if (collectionIds) { scope = ` AND vr.collection_id IN (SELECT value FROM json_each(?))`; @@ -4664,10 +4688,7 @@ function metadataEligibleVectorRows(db: Database, filter: MetadataFilter, collec JOIN document_metadata dm ON dm.document_id = d.id JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.name = d.collection JOIN ${VEC_ROWS_TABLE} vr ON vr.hash = d.hash AND vr.collection_id = ci.id - WHERE d.active = 1 - AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} - AND dm.extraction_error IS NULL - AND ${compiled.sql}${scope} + WHERE d.active = 1 AND ${current.sql}${scope} `).all(...params) as { rowid: number; collectionId: number }[]; const byCollection = new Map(); for (const row of rows) { @@ -4678,6 +4699,71 @@ function metadataEligibleVectorRows(db: Database, filter: MetadataFilter, collec return byCollection; } +/** + * Prepares step 2 of a vector search: the documents behind a set of vector + * rows, one per file at its nearest chunk, nearest first. The rowids are bound + * as one JSON parameter, so the statement text stays fixed and a match set up + * to the 4096 k cap never meets SQLite's bound-parameter limit. Bodies are + * left out: one long document can hold every match, and only the final results + * load theirs. + */ +function vecDocumentResolver(db: Database, filter?: MetadataFilter): (matches: readonly VecMatch[]) => VecDocumentMatch[] { + // Re-apply the filter on the document join: vectors are content-scoped, + // so one hash can belong to both matching and non-matching documents. + const current = filter ? compileCurrentMetadataFilter(filter) : undefined; + const statement = withLazyContentVectorMigration(db, () => db.prepare(` + SELECT + vr.id AS rowid, + cv.hash, + cv.pos, + 'qmd://' || d.collection || '/' || d.path as filepath, + d.collection || '/' || d.path as display_path, + d.title, + dm.metadata_json + FROM json_each(?) j + JOIN ${VEC_ROWS_TABLE} vr ON vr.id = j.value + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + JOIN content_vectors cv ON cv.hash = vr.hash AND cv.seq = vr.seq + JOIN documents d ON d.hash = vr.hash AND d.collection = ci.name + JOIN content ON content.hash = d.hash + LEFT JOIN document_metadata dm ON dm.document_id = d.id + WHERE d.active = 1${current ? ` AND ${current.sql}` : ""} + `)); + return (matches) => { + if (matches.length === 0) return []; + const distanceByRowid = new Map(matches.map(r => [r.rowid, r.distance])); + const rows = statement.all(rowidList(matches.map(r => r.rowid)), ...(current?.params ?? [])) as Omit[]; + const best = new Map(); + for (const row of rows) { + const distance = distanceByRowid.get(row.rowid) ?? 1; + const existing = best.get(row.filepath); + if (!existing || distance < existing.distance) best.set(row.filepath, { ...row, distance }); + } + return Array.from(best.values()).sort((a, b) => a.distance - b.distance); + }; +} + +/** + * Nearest documents of one scan target. The first KNN asks for three chunks + * per requested document. Chunks of one long document can still fill every + * slot, so while the matches collapse into fewer than `limit` documents and + * the target holds rows beyond them, k doubles, up to sqlite-vec's cap. + */ +function nearestVecDocuments( + scan: ReturnType, + resolve: ReturnType, + queryVec: Float32Array, + limit: number, + target: VecScanTarget, +): VecDocumentMatch[] { + for (let k = limit * 3; ; k *= 2) { + const vecK = Math.max(1, Math.min(SQLITE_VEC_MAX_K, k)); + const matches = scan(queryVec, vecK, target); + const documents = resolve(matches); + if (documents.length >= limit || matches.length < vecK || vecK === SQLITE_VEC_MAX_K) return documents; + } +} + export async function searchVec(db: Database, query: string, model: string, limit: number = 20, collectionName?: string | readonly string[], session?: ILLMSession, precomputedEmbedding?: number[], llm?: LlamaCpp, filter?: MetadataFilter): Promise { if (!hasVectorIndex(db)) return []; @@ -4697,72 +4783,29 @@ export async function searchVec(db: Database, query: string, model: string, limi // "optimize" this by combining into a single query with JOINs - it will break. // See: https://github.com/tobi/qmd/pull/23 - // Step 1: Get vector matches from sqlite-vec (no JOINs allowed): one KNN per - // collection in scope, three candidate chunks per requested result, so - // multi-chunk documents can still yield `limit` unique files. One statement - // per member rather than `collection_id IN (...)`: the IN form yields k rows - // per value only because SQLite runs vec0's filter once per value, which is - // a planner detail rather than a vec0 contract. + // Step 1 gets vector matches from sqlite-vec (no JOINs allowed), one KNN per + // collection in scope; step 2 resolves them to documents. One statement per + // member rather than `collection_id IN (...)`: the IN form yields k rows per + // value only because SQLite runs vec0's filter once per value, which is a + // planner detail rather than a vec0 contract. const targets: VecScanTarget[] = collectionIds ? collectionIds.map(collectionId => ({ collectionId, eligibleRowids: eligible?.get(collectionId) })) : [{ eligibleRowids: eligible && Array.from(eligible.values()).flat() }]; const scanTargets = eligible ? targets.filter(t => t.eligibleRowids?.length) : targets; if (scanTargets.length === 0) return []; const scan = knnVecScanner(db, collectionIds !== undefined, eligible !== undefined); + const resolve = vecDocumentResolver(db, filter); const queryVec = new Float32Array(embedding); - const vecResults = scanTargets.flatMap(target => scan(queryVec, limit * 3, target)); - if (vecResults.length === 0) return []; - - // Step 2: Get chunk info and document data by rowid. The rowids are bound - // as one JSON parameter: thirteen partitions at the k cap exceed SQLite's - // limit on separate bound parameters. - const docParams: SQLiteValue[] = [rowidList(vecResults.map(r => r.rowid))]; - let docFilter = ""; - if (filter) { - // Re-apply the filter on the document join: vectors are content-scoped, - // so one hash can belong to both matching and non-matching documents. - const compiled = compileMetadataFilter(filter, "d"); - docFilter = ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiled.sql}`; - docParams.push(...compiled.params); - } - const distanceByRowid = new Map(vecResults.map(r => [r.rowid, r.distance])); - const docRows = withLazyContentVectorMigration(db, () => db.prepare(` - SELECT - vr.id AS rowid, - cv.hash, - cv.pos, - 'qmd://' || d.collection || '/' || d.path as filepath, - d.collection || '/' || d.path as display_path, - d.title, - ${cappedBodySql("content.doc")} as body, - dm.metadata_json - FROM json_each(?) j - JOIN ${VEC_ROWS_TABLE} vr ON vr.id = j.value - JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id - JOIN content_vectors cv ON cv.hash = vr.hash AND cv.seq = vr.seq - JOIN documents d ON d.hash = vr.hash AND d.collection = ci.name - JOIN content ON content.hash = d.hash - LEFT JOIN document_metadata dm ON dm.document_id = d.id - WHERE d.active = 1${docFilter} - `).all(...docParams) as { - rowid: number; hash: string; pos: number; filepath: string; - display_path: string; title: string; body: string; metadata_json: string | null; - }[]); - - // Combine with distances and dedupe by filepath - const seen = new Map(); - for (const row of docRows) { - const distance = distanceByRowid.get(row.rowid) ?? 1; - const existing = seen.get(row.filepath); - if (!existing || distance < existing.bestDist) { - seen.set(row.filepath, { row, bestDist: distance }); - } - } + const bodyOf = db.prepare(`SELECT doc FROM content WHERE hash = ?`); - return Array.from(seen.values()) - .sort((a, b) => a.bestDist - b.bestDist) + // Each target yields its own nearest `limit` documents (or all it holds), so + // merging them by distance gives the scope's exact nearest `limit`. + return scanTargets + .flatMap(target => nearestVecDocuments(scan, resolve, queryVec, limit, target)) + .sort((a, b) => a.distance - b.distance) .slice(0, limit) - .map(({ row, bestDist }) => { + .map((row) => { + const body = (bodyOf.get(row.hash) as { doc: string }).doc; const collectionName = row.filepath.split('//')[1]?.split('/')[0] || ""; return { filepath: row.filepath, @@ -4772,11 +4815,11 @@ export async function searchVec(db: Database, query: string, model: string, limi docid: getDocid(row.hash), collectionName, modifiedAt: "", // Not available in vec query - bodyLength: row.body.length, - body: row.body, + bodyLength: body.length, + body, context: getContextForFile(db, row.filepath), metadata: parseMetadataJson(row.metadata_json), - score: 1 - bestDist, // Cosine similarity = 1 - cosine distance + score: 1 - row.distance, // Cosine similarity = 1 - cosine distance source: "vec" as const, chunkPos: row.pos, }; diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index 9d973d086..9c0595628 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -284,6 +284,38 @@ describe("searchVec with metadata filter", () => { expect(filtered.map(r => r.displayPath)).toEqual(["docs/b.md"]); }); + const eligibleOnly: MetadataFilter = { key: "eligible", operator: "eq", value: true }; + + /** An eligible document and an ineligible copy of its content in one collection, one chunk per vector. */ + async function insertLongDocumentWithExcludedCopy(vectors: number[][]): Promise { + const body = "# Many chunks"; + const { hash } = await insertDoc("book", "many-chunks.md", body, { eligible: true }); + await insertDoc("book", "excluded-copy.md", body, { eligible: false }); + const now = new Date().toISOString(); + vectors.forEach((vector, seq) => insertEmbedding(store.db, hash, seq, seq * 100, new Float32Array(vector), model, now, vectors.length)); + } + + test("a document with many close chunks does not starve another eligible document", async () => { + store.ensureVecTable(3); + await insertLongDocumentWithExcludedCopy(Array.from({ length: 20 }, () => [1, 0, 0])); + await insertEmbeddedDoc("book", "second-document.md", "# Second document", [0, 1, 0], { eligible: true }); + + const results = await searchVec(store.db, "q", model, 2, undefined, undefined, queryEmbedding, undefined, eligibleOnly); + expect(results.map(r => r.displayPath)).toEqual(["book/many-chunks.md", "book/second-document.md"]); + }); + + test.each([20, 450])("the best chunk of a %i-chunk document survives deduplication", async (chunks) => { + store.ensureVecTable(3); + await insertLongDocumentWithExcludedCopy( + Array.from({ length: chunks }, (_, seq) => (seq === chunks - 1 ? [1, 0, 0] : [0, 0, 1])), + ); + await insertEmbeddedDoc("book", "second-document.md", "# Second document", [-1, 0.1, 0], { eligible: true }); + + const results = await searchVec(store.db, "q", model, 2, undefined, undefined, queryEmbedding, undefined, eligibleOnly); + expect(results.map(r => r.displayPath)).toEqual(["book/many-chunks.md", "book/second-document.md"]); + expect(results[0]!.chunkPos).toBe((chunks - 1) * 100); + }); + test("a filter admitting more than 20,000 chunks still returns its nearest eligible documents", async () => { store.ensureVecTable(3); store.db.exec("BEGIN"); @@ -305,20 +337,10 @@ describe("searchVec with metadata filter", () => { test("shared content hash within one collection returns only the matching document path", async () => { store.ensureVecTable(3); - const now = new Date().toISOString(); const body = "# Shared body"; - const hash = await hashContent(body); - insertContent(store.db, hash, body, now); - - const publishedId = insertDocument(store.db, "notes", "published-copy.md", "t", hash, now, now); - const draftId = insertDocument(store.db, "notes", "draft-copy.md", "t", hash, now, now); - replaceDocumentMetadata(store.db, publishedId, { - metadata: { status: "published" }, extractionVersion: METADATA_EXTRACTION_VERSION, - }); - replaceDocumentMetadata(store.db, draftId, { - metadata: { status: "draft" }, extractionVersion: METADATA_EXTRACTION_VERSION, - }); - insertEmbedding(store.db, hash, 0, 0, new Float32Array([1, 0, 0]), model, now, 1); + const { hash } = await insertDoc("notes", "published-copy.md", body, { status: "published" }); + await insertDoc("notes", "draft-copy.md", body, { status: "draft" }); + insertEmbedding(store.db, hash, 0, 0, new Float32Array([1, 0, 0]), model, new Date().toISOString(), 1); const filtered = await searchVec( store.db, "q", model, 10, "notes", undefined, queryEmbedding, undefined, diff --git a/test/store.test.ts b/test/store.test.ts index a19634daf..c716cffa6 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -4744,6 +4744,23 @@ describe("Vector Search collection filter", () => { await cleanupTestDb(store); }); + test("searchVec fills its limit when one document's chunks take every candidate slot", async () => { + const store = await createTestStore(); + const book = await createTestCollection({ name: "book", pwd: "/test/book" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, book, "manychunks", Array.from({ length: 20 }, () => vector(1, 0))); + await insertVecDoc(store, book, "seconddoc", [vector(0.2, 1)]); + + for (const scope of [book, undefined]) { + for (const limit of [2, 5]) { + const results = await store.searchVec("ignored", "test-model", limit, scope, undefined, query); + expect(results.map((r) => r.hash)).toEqual(["manychunks", "seconddoc"]); + } + } + + await cleanupTestDb(store); + }); + test("searchVec rows carry the vec source, a cosine score, and the best chunk position", async () => { const store = await createTestStore(); const embedded = await createTestCollection({ name: "embedded", pwd: "/test/embedded" }); @@ -6285,10 +6302,10 @@ describe("Embedding batching", () => { expect(db.prepare(`SELECT COUNT(*) AS c FROM documents WHERE active = 1`).get()).toEqual({ c: 2 }); expect(db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get()).toEqual({ c: 6 }); - // Four stale rows sit nearer the query than both live documents: with - // limit 1 every one of the scoped KNN's three slots goes to a row no - // active document owns, and the search comes back empty. - expect(await store.searchVec("ignored", "test-model", 1, "docs", undefined, near)).toEqual([]); + // Four stale rows sit nearer the query than both live documents and take + // every one of the first scoped KNN's three slots at limit 1; the search + // only answers after widening its KNN past them. + expect(await store.searchVec("ignored", "test-model", 1, "docs", undefined, near)).toHaveLength(1); expect(cleanupOrphanedVectors(db)).toBe(4); From 5807d234221474d3acd57629a2cb4b22b23f8d81 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 24 Sep 2026 21:55:55 -0500 Subject: [PATCH 37/82] fix(update): copy vectors into gained collections before removing stale rows `qmd update` and the library's `update()` removed stale partition rows first and ran the copy pass second. When a file moved from one collection to another between two updates, the cleanup deleted the old collection's partition rows for the hash, and with them its content_vectors rows, before the copy could read them; the copy then found no source, queued the hash, and the moved document needed a fresh embed. The copy now runs first and reads the rows the hash is leaving, and the cleanup afterwards removes only the stale row. The SDK suite gains a test that moves an embedded file into a second collection and checks that its content_vectors rows survive, its only partition row is in the new collection, and nothing needs embedding; it failed with 0 of 1 content_vectors rows before the change. The CHANGELOG entry now says what the cleanup removes: the vectors of content no indexed document holds any more, which a later return of that content embeds again. (cherry picked from commit bd86d162e70c0c6bedf6e53f40734f88a2d363b7) --- CHANGELOG.md | 17 +++++++------- src/cli/qmd.ts | 10 +++++---- src/index.ts | 6 +++-- test/sdk.test.ts | 58 +++++++++++++++++++++++++++++++++++++++++++++++- 4 files changed, 76 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 297a7214b..7e742ee83 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -97,14 +97,15 @@ progress; the conversion resumes from where it stopped if interrupted, two commands started at once share it, and a VACUUM at the end reclaims the space of the old table. `qmd collection rename` no longer touches vectors, - `qmd update` removes the vector rows of changed files as it finishes, and - `qmd embed` copies an already-embedded document into a collection that - gains it instead of embedding it again. A metadata-filtered vector search - returns the nearest documents the filter admits however many chunks it - admits, and a vector search fills its limit even when one long document - holds all of the nearest chunks. Downgrading to an older qmd afterwards - needs `qmd embed -f`, and a `qmd mcp` server started before the upgrade - must be restarted. + `qmd update` removes, as it finishes, the vectors of content that no + indexed document holds any more (content that comes back later is embedded + again), and `qmd update` and `qmd embed` copy an already-embedded document + into a collection that gains it instead of embedding it again. A + metadata-filtered vector search returns the nearest documents the filter + admits however many chunks it admits, and a vector search fills its limit + even when one long document holds all of the nearest chunks. Downgrading + to an older qmd afterwards needs `qmd embed -f`, and a `qmd mcp` server + started before the upgrade must be restarted. ## [2.8.3] - 2026-08-16 diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index ddeb20b3f..f1297bf03 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -1017,15 +1017,17 @@ async function updateCollections(): Promise { console.log(""); } + // The pending count below only sees content_vectors, so a hash that joined + // a collection while already embedded elsewhere would be neither counted + // nor searchable there; copying its rows closes that gap without a model. + // The copy runs before the cleanup, which deletes the partition rows the + // copy reads from, so a document moved between collections keeps its vectors. + const copiedVectors = copyVectorsToNewCollections(db).copied; // A changed file rewrites its document's hash in place, which strands the // old hash's rows in the collection's vector partition; the partition // filter cannot see documents.active, so those rows would take k slots // from a scoped search until they are removed. const staleVectors = cleanupOrphanedVectors(db); - // The pending count below only sees content_vectors, so a hash that joined - // a collection while already embedded elsewhere would be neither counted - // nor searchable there; copying its rows closes that gap without a model. - const copiedVectors = copyVectorsToNewCollections(db).copied; // Check if any documents need embedding (show once at end) const needsEmbedding = getHashesNeedingEmbedding(db); diff --git a/src/index.ts b/src/index.ts index e3b08afe9..76bc2a625 100644 --- a/src/index.ts +++ b/src/index.ts @@ -630,9 +630,11 @@ export async function createStore(options: StoreOptions): Promise { // A changed file rewrites its document's hash in place and strands the // old hash's partition rows, which take k slots from scoped searches; // a hash that joined a collection while embedded elsewhere stays - // unsearchable there until its rows are copied. - const staleVectorsRemoved = cleanupOrphanedVectors(db); + // unsearchable there until its rows are copied. The copy runs first: + // the cleanup deletes the partition rows it copies from, so a document + // moved between collections would otherwise need a fresh embed. const vectorsCopied = copyVectorsToNewCollections(db).copied; + const staleVectorsRemoved = cleanupOrphanedVectors(db); return { collections: filtered.length, diff --git a/test/sdk.test.ts b/test/sdk.test.ts index 84bff242d..de9af83a1 100644 --- a/test/sdk.test.ts +++ b/test/sdk.test.ts @@ -6,7 +6,7 @@ */ import { describe, test, expect, beforeAll, afterAll, beforeEach, afterEach } from "vitest"; -import { mkdtemp, writeFile, mkdir, rm } from "node:fs/promises"; +import { mkdtemp, writeFile, mkdir, rm, rename } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { existsSync, writeFileSync, mkdirSync, readFileSync } from "node:fs"; @@ -1158,6 +1158,62 @@ describe("embed", () => { } }); + test("store.update keeps the vectors of a document moved to another collection", async () => { + const fromDir = join(testDir, `vector-move-from-${Date.now()}`); + const toDir = join(testDir, `vector-move-to-${Date.now()}`); + await mkdir(fromDir, { recursive: true }); + await mkdir(toDir, { recursive: true }); + await writeFile(join(fromDir, "moved.md"), "# Moved\n\nEmbedded in one collection, then moved to another.\n"); + + const store = await createStore({ + dbPath: freshDbPath(), + config: { + collections: { + from: { path: fromDir, pattern: "**/*.md" }, + to: { path: toDir, pattern: "**/*.md" }, + }, + }, + }); + setDefaultLlamaCpp(createFakeTokenizer() as any); + store.internal.llm = createFakeEmbedLlm() as any; + + const db = store.internal.db; + const hashOf = () => + (db.prepare(`SELECT hash FROM documents WHERE path = 'moved.md' LIMIT 1`).get() as { hash: string }).hash; + const chunkCount = (hash: string) => + (db.prepare(`SELECT COUNT(*) AS n FROM content_vectors WHERE hash = ?`).get(hash) as { n: number }).n; + const partitionsOf = (hash: string): string[] => + (db.prepare(` + SELECT ci.name AS collection FROM ${VEC_ROWS_TABLE} vr + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + WHERE vr.hash = ? + ORDER BY ci.name + `).all(hash) as Array<{ collection: string }>).map((row) => row.collection); + + try { + await store.update(); + await store.embed(); + const hash = hashOf(); + const embeddedChunks = chunkCount(hash); + expect(embeddedChunks).toBeGreaterThan(0); + expect(partitionsOf(hash)).toEqual(["from"]); + + await rename(join(fromDir, "moved.md"), join(toDir, "moved.md")); + const result = await store.update(); + + expect(result.removed).toBe(1); + expect(result.indexed).toBe(1); + expect(chunkCount(hash)).toBe(embeddedChunks); + expect(partitionsOf(hash)).toEqual(["to"]); + expect(result.vectorsCopied).toBe(1); + expect(result.staleVectorsRemoved).toBe(1); + expect(result.needsEmbedding).toBe(0); + } finally { + setDefaultLlamaCpp(null); + await store.close(); + } + }); + test("store.embed rejects invalid batch limits", async () => { const store = await createStore({ dbPath: freshDbPath(), From cd4723dc7a9d513f0a632cc0c20641a009aee2a2 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 24 Sep 2026 21:55:55 -0500 Subject: [PATCH 38/82] fix(store): make collection rename and vector index drops atomic `renameCollection` ran its store_collections, documents and vector-id updates as separate statements. `vector_collection_ids.name` is unique, and a collection removed while sqlite-vec is not loaded keeps its id, so renaming onto that name failed after the documents had already moved, leaving them under a name whose partition id still belonged to the old collection. The rename now runs in one IMMEDIATE transaction: it renames the store_collections row first (so a name that already exists fails before anything moves), drops a leftover partition and id under the target name, refuses with an explicit message when that id cannot be dropped without sqlite-vec, then moves the documents and the id. The unscoped `clearAllEmbeddings`, the scoped clear that empties the index, and `ensureVecTable`'s dimension reset each emptied `vector_rows` and dropped the vec0 table in separate statements. `vector_rows` ids are reused once the map is empty, so a process stopped between the two left vec0 rows whose rowids new inserts then collided with. One helper now drops the table and empties the map in a single commit, and the unscoped clear deletes content_vectors in the same transaction. `searchVec` reads each final result's body after resolving the documents, so another process's orphaned-content cleanup can delete that row in between; the search now drops that result instead of throwing. Tests cover a rename onto a leftover id with and without a droppable partition, a rename onto an existing collection, a failure injected on the last rename step (all names unchanged), a vanished content row during a search, and a DROP forced to fail during a clear and during a dimension reset (nothing committed). Each failed before the change. (cherry picked from commit 8db74fa5c453e66c6b16b9a0ef7b493ea185e46f) --- src/store.ts | 78 ++++++++++++------ test/store.test.ts | 197 ++++++++++++++++++++++++++++++++++++++++++++- 2 files changed, 248 insertions(+), 27 deletions(-) diff --git a/src/store.ts b/src/store.ts index d66584e5d..4a031ecc4 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1540,6 +1540,19 @@ export function isSqliteVecAvailable(): boolean { return _sqliteVecAvailable === true; } +/** + * Drop the vec0 table and empty its rowid map in one commit. vector_rows ids + * are reused once the map is empty, so a map emptied without the drop hands + * new rows rowids that surviving vec0 rows still hold, and every later insert + * fails on the vec0 primary key. + */ +function dropVectorIndex(db: Database): void { + db.transaction(() => { + db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); + db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); + }).immediate(); +} + function ensureVecTableInternal(db: Database, dimensions: number): void { if (!_sqliteVecAvailable) { throw createSqliteVecUnavailableError( @@ -1559,8 +1572,7 @@ function ensureVecTableInternal(db: Database, dimensions: number): void { `Run 'qmd embed -f' to re-embed with the new model.` ); } - db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); - db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); + dropVectorIndex(db); } createPartitionedVecTable(db, dimensions); } @@ -4118,21 +4130,35 @@ export function removeCollection(db: Database, collectionName: string): { delete } /** - * Rename a collection. - * Updates both YAML config and database documents table. + * Rename a collection: its documents, its vector partition id, and its + * store_collections row. */ export function renameCollection(db: Database, oldName: string, newName: string): void { - // Update all documents with the new collection name in database - db.prepare(`UPDATE documents SET collection = ? WHERE collection = ?`) - .run(newName, oldName); - // The documents keep their ids and paths, so their sync rows stay valid - // under the new name. Rows already under it belong to no collection. - db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(newName); - db.prepare(`UPDATE file_sync_state SET collection = ? WHERE collection = ?`).run(newName, oldName); - renameCollectionId(db, oldName, newName); + // One commit: the vector id name is UNIQUE, so a rename that fails after + // the documents moved would leave them under a name whose partition id + // still belongs to the old one, and vector search would miss them. + db.transaction(() => { + renameStoreCollection(db, oldName, newName); + + // A removed collection keeps its id while sqlite-vec is not loaded to + // drop its partition; the renamed collection's id cannot take that name. + if (resolveCollectionId(db, newName) !== undefined) { + deleteVectorPartition(db, newName); + if (resolveCollectionId(db, newName) !== undefined) { + throw new Error( + `Cannot rename to '${newName}': a removed collection of that name still has a vector partition, which only qmd with sqlite-vec loaded can drop. ` + + `Open qmd with sqlite-vec loaded and retry the rename.` + ); + } + } - // Rename in store_collections - renameStoreCollection(db, oldName, newName); + db.prepare(`UPDATE documents SET collection = ? WHERE collection = ?`).run(newName, oldName); + // The documents keep their ids and paths, so their sync rows stay valid + // under the new name. Rows already under it belong to no collection. + db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(newName); + db.prepare(`UPDATE file_sync_state SET collection = ? WHERE collection = ?`).run(newName, oldName); + renameCollectionId(db, oldName, newName); + }).immediate(); } // ============================================================================= @@ -4804,10 +4830,14 @@ export async function searchVec(db: Database, query: string, model: string, limi .flatMap(target => nearestVecDocuments(scan, resolve, queryVec, limit, target)) .sort((a, b) => a.distance - b.distance) .slice(0, limit) - .map((row) => { - const body = (bodyOf.get(row.hash) as { doc: string }).doc; + .flatMap((row): SearchResult[] => { + // The body is read after resolution, outside its snapshot: another + // process's orphaned-content cleanup can delete the row in between. + const content = bodyOf.get(row.hash) as { doc: string } | null | undefined; + if (content == null) return []; + const body = content.doc; const collectionName = row.filepath.split('//')[1]?.split('/')[0] || ""; - return { + return [{ filepath: row.filepath, displayPath: row.display_path, title: row.title, @@ -4822,7 +4852,7 @@ export async function searchVec(db: Database, query: string, model: string, limi score: 1 - row.distance, // Cosine similarity = 1 - cosine distance source: "vec" as const, chunkPos: row.pos, - }; + }]; }); } @@ -4887,9 +4917,10 @@ export function getHashesForEmbedding(db: Database, model: string = DEFAULT_EMBE */ export function clearAllEmbeddings(db: Database, collection?: string): void { if (!collection) { - db.exec(`DELETE FROM content_vectors`); - db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); - db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); + db.transaction(() => { + db.exec(`DELETE FROM content_vectors`); + dropVectorIndex(db); + }).immediate(); return; } @@ -4923,10 +4954,7 @@ export function clearAllEmbeddings(db: Database, collection?: string): void { const remaining = db .prepare(`SELECT COUNT(*) AS n FROM content_vectors`) .get() as { n: number }; - if (remaining.n === 0) { - db.exec(`DELETE FROM ${VEC_ROWS_TABLE}`); - db.exec(`DROP TABLE IF EXISTS ${VEC_TABLE}`); - } + if (remaining.n === 0) dropVectorIndex(db); }).immediate()); } diff --git a/test/store.test.ts b/test/store.test.ts index c716cffa6..ca8cb07be 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -8,7 +8,7 @@ import { describe, test, expect, beforeAll, afterAll, beforeEach, afterEach, vi } from "vitest"; import { openDatabase, loadSqliteVec, isBun } from "../src/db.js"; -import type { Database } from "../src/db.js"; +import type { Database, SQLiteValue } from "../src/db.js"; import { unlink, mkdtemp, rmdir, writeFile, rm, mkdir, rename, chmod, readFile, symlink, utimes } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; @@ -70,6 +70,7 @@ import { _resetProductionModeForTesting, hybridQuery, structuredSearch, + searchVec, vectorSearchQuery, type Store, type DocumentResult, @@ -78,7 +79,7 @@ import { type RankedListMeta, } from "../src/store.js"; import type { CollectionConfig } from "../src/collections.js"; -import { VEC_ROWS_TABLE, VEC_TABLE, deletePartitionRows, resolveCollectionId, vecInteger } from "../src/vec-layout.js"; +import { LEGACY_VEC_TABLE, VEC_COLLECTION_IDS_TABLE, VEC_ROWS_TABLE, VEC_TABLE, deletePartitionRows, resolveCollectionId, vecInteger } from "../src/vec-layout.js"; // ============================================================================= // LlamaCpp Setup @@ -4195,6 +4196,31 @@ describe("Vector Table", () => { await cleanupTestDb(store); }); + + test("ensureVecTable keeps a dimensionless vector table and its row map when clearing the rows fails", async () => { + const store = await createTestStore(); + try { + // No float[N] in the declaration, so ensureVecTable drops and recreates it. + const dimensionless = `CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`; + store.db.exec(dimensionless); + store.db.prepare(`INSERT INTO ${VEC_ROWS_TABLE} (hash, seq, collection_id) VALUES ('h1', 0, 1)`).run(); + store.db.exec(`CREATE TRIGGER block_vector_rows_delete BEFORE DELETE ON ${VEC_ROWS_TABLE} BEGIN SELECT RAISE(ABORT, 'injected failure clearing vector rows'); END`); + const vecTableSql = () => (store.db.prepare(`SELECT sql FROM sqlite_master WHERE name = ?`).get(VEC_TABLE) as { sql: string } | null | undefined)?.sql; + const rowCount = () => (store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get() as { c: number }).c; + + expect(() => store.ensureVecTable(8)).toThrow("injected failure clearing vector rows"); + + expect(vecTableSql()).toBe(dimensionless); + expect(rowCount()).toBe(1); + + store.db.exec(`DROP TRIGGER block_vector_rows_delete`); + store.ensureVecTable(8); + expect(vecTableSql()).toContain("float[8]"); + expect(rowCount()).toBe(0); + } finally { + await cleanupTestDb(store); + } + }); }); // ============================================================================= @@ -4889,6 +4915,148 @@ describe("Vector Search collection filter", () => { await cleanupTestDb(store); }); + test("searchVec drops a result whose content row vanishes after its document resolved", async () => { + const store = await createTestStore(); + const collection = await createTestCollection({ name: "racing", pwd: "/test/racing" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, collection, "nearhash", [vector(1, 0)]); + await insertVecDoc(store, collection, "vanishhash", [vector(0.9, 0.44)]); + await insertVecDoc(store, collection, "farhash", [vector(0.8, 0.6)]); + + // Replays another process's orphaned-content cleanup landing between + // document resolution and the body read. + const bodySql = "SELECT doc FROM content WHERE hash = ?"; + const racing: Database = { + prepare: (sql: string) => { + const statement = store.db.prepare(sql); + if (!sql.includes(bodySql)) return statement; + return { + run: (...params: SQLiteValue[]) => statement.run(...params), + all: (...params: SQLiteValue[]) => statement.all(...params), + iterate: (...params: SQLiteValue[]) => statement.iterate(...params), + get: (...params: SQLiteValue[]) => { + if (params[0] === "vanishhash") store.db.prepare(`DELETE FROM content WHERE hash = ?`).run("vanishhash"); + return statement.get(...params); + }, + }; + }, + transaction: (fn) => store.db.transaction(fn), + exec: (sql: string) => store.db.exec(sql), + loadExtension: (path: string) => store.db.loadExtension(path), + close: () => store.db.close(), + }; + + const results = await searchVec(racing, "ignored", "test-model", 3, undefined, undefined, query); + + expect(results.map((r) => r.hash)).toEqual(["nearhash", "farhash"]); + expect(results.map((r) => r.body)).toEqual(["Document nearhash", "Document farhash"]); + + await cleanupTestDb(store); + }); + + function documentsIn(store: Store, collection: string): number { + return (store.db.prepare(`SELECT COUNT(*) AS c FROM documents WHERE collection = ?`).get(collection) as { c: number }).c; + } + + function storeCollectionNames(store: Store, ...names: string[]): string[] { + return (store.db.prepare(`SELECT name FROM store_collections WHERE name IN (${names.map(() => "?").join(", ")}) ORDER BY name`) + .all(...names) as { name: string }[]).map((row) => row.name); + } + + test("renameCollection onto a removed collection's leftover partition drops the leftover and keeps the renamed vectors", async () => { + const store = await createTestStore(); + const before = await createTestCollection({ name: "before", pwd: "/test/before" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, before, "renamedhash", [vector(1, 0)]); + const beforeId = resolveCollectionId(store.db, before)!; + // A removed collection whose partition and id outlived its documents. + await insertVecDoc(store, "after", "leftoverhash", [vector(0.9, 0.44), vector(0.8, 0.6)]); + store.db.prepare(`DELETE FROM documents WHERE collection = 'after'`).run(); + expect(vectorRowCount(store, "after")).toBe(2); + + renameCollection(store.db, before, "after"); + + const renamed = await store.searchVec("ignored", "test-model", 3, "after", undefined, query); + expect(renamed.map((r) => r.hash)).toEqual(["renamedhash"]); + expect(renamed[0]!.collectionName).toBe("after"); + expect(resolveCollectionId(store.db, "after")).toBe(beforeId); + expect(resolveCollectionId(store.db, before)).toBeUndefined(); + expect(vectorRowCount(store, "after")).toBe(1); + expect((store.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE}`).get() as { c: number }).c).toBe(1); + expect(documentsIn(store, "after")).toBe(1); + expect(storeCollectionNames(store, before, "after")).toEqual(["after"]); + + await cleanupTestDb(store); + }); + + test("renameCollection rolls back every table when a later step fails", async () => { + const store = await createTestStore(); + const before = await createTestCollection({ name: "before", pwd: "/test/before" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, before, "renamedhash", [vector(1, 0)]); + const beforeId = resolveCollectionId(store.db, before)!; + + const failing = failingDb(store.db, `UPDATE ${VEC_COLLECTION_IDS_TABLE} SET name = ?`, "injected failure renaming the vector id"); + expect(() => renameCollection(failing, before, "after")).toThrow("injected failure renaming the vector id"); + + expect(documentsIn(store, before)).toBe(1); + expect(documentsIn(store, "after")).toBe(0); + expect(resolveCollectionId(store.db, before)).toBe(beforeId); + expect(resolveCollectionId(store.db, "after")).toBeUndefined(); + expect(storeCollectionNames(store, before, "after")).toEqual([before]); + expect((await store.searchVec("ignored", "test-model", 3, before, undefined, query)).map((r) => r.hash)).toEqual(["renamedhash"]); + + // The connection is left clean: a plain retry renames everything. + renameCollection(store.db, before, "after"); + expect(documentsIn(store, "after")).toBe(1); + expect(resolveCollectionId(store.db, "after")).toBe(beforeId); + expect(storeCollectionNames(store, before, "after")).toEqual(["after"]); + + await cleanupTestDb(store); + }); + + test("renameCollection onto an existing collection moves nothing and keeps the target's vectors", async () => { + const store = await createTestStore(); + const before = await createTestCollection({ name: "before", pwd: "/test/before" }); + const taken = await createTestCollection({ name: "taken", pwd: "/test/taken" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, before, "renamedhash", [vector(1, 0)]); + await insertVecDoc(store, taken, "takenhash", [vector(0.9, 0.44)]); + const takenId = resolveCollectionId(store.db, taken)!; + + expect(() => renameCollection(store.db, before, taken)).toThrow(`Collection '${taken}' already exists`); + + expect(documentsIn(store, before)).toBe(1); + expect(documentsIn(store, taken)).toBe(1); + expect(resolveCollectionId(store.db, taken)).toBe(takenId); + expect(vectorRowCount(store, taken)).toBe(1); + expect((await store.searchVec("ignored", "test-model", 3, taken, undefined, query)).map((r) => r.hash)).toEqual(["takenhash"]); + + await cleanupTestDb(store); + }); + + test("renameCollection refuses a target whose leftover id cannot be dropped yet and changes nothing", async () => { + const store = await createTestStore(); + const before = await createTestCollection({ name: "before", pwd: "/test/before" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, before, "renamedhash", [vector(1, 0)]); + const beforeId = resolveCollectionId(store.db, before)!; + store.db.prepare(`INSERT INTO ${VEC_COLLECTION_IDS_TABLE} (name) VALUES ('after')`).run(); + const leftoverId = resolveCollectionId(store.db, "after")!; + // A legacy table awaiting migration: partitions are not dropped until it runs. + store.db.exec(`CREATE VIRTUAL TABLE ${LEGACY_VEC_TABLE} USING vec0(hash_seq TEXT PRIMARY KEY, embedding float[${DIMS}] distance_metric=cosine)`); + + expect(() => renameCollection(store.db, before, "after")).toThrow(/'after'.*sqlite-vec/s); + + expect(documentsIn(store, before)).toBe(1); + expect(documentsIn(store, "after")).toBe(0); + expect(resolveCollectionId(store.db, before)).toBe(beforeId); + expect(resolveCollectionId(store.db, "after")).toBe(leftoverId); + expect(storeCollectionNames(store, before, "after")).toEqual([before]); + + await cleanupTestDb(store); + }); + test("removeCollection drops the collection's partition and leaves the others", async () => { const store = await createTestStore(); const gone = await createTestCollection({ name: "gone", pwd: "/test/gone" }); @@ -4961,6 +5129,31 @@ describe("Vector Search collection filter", () => { await cleanupTestDb(store); }); + + test("clearAllEmbeddings for the whole index rolls back every table when the vec0 drop fails", async () => { + const store = await createTestStore(); + const collection = await createTestCollection({ name: "cleared", pwd: "/test/cleared" }); + store.ensureVecTable(DIMS); + await insertVecDoc(store, collection, "firsthash", [vector(1, 0), vector(0.9, 0.44)]); + await insertVecDoc(store, collection, "secondhash", [vector(0.8, 0.6)]); + const count = (table: string) => (store.db.prepare(`SELECT COUNT(*) AS c FROM ${table}`).get() as { c: number }).c; + const counts = () => ({ contentVectors: count("content_vectors"), rows: count(VEC_ROWS_TABLE), vectors: count(VEC_TABLE) }); + expect(counts()).toEqual({ contentVectors: 3, rows: 3, vectors: 3 }); + + const failing = failingDb(store.db, `DROP TABLE IF EXISTS ${VEC_TABLE}`, "injected failure dropping the vec0 table"); + expect(() => clearAllEmbeddings(failing)).toThrow("injected failure dropping the vec0 table"); + + expect(counts()).toEqual({ contentVectors: 3, rows: 3, vectors: 3 }); + expect((await store.searchVec("ignored", "test-model", 5, collection, undefined, query)).map((r) => r.hash)).toEqual(["firsthash", "secondhash"]); + + // The connection is left clean: a plain retry clears everything. + clearAllEmbeddings(store.db); + expect(count("content_vectors")).toBe(0); + expect(count(VEC_ROWS_TABLE)).toBe(0); + expect(store.db.prepare(`SELECT name FROM sqlite_master WHERE name = ?`).get(VEC_TABLE)).toBeFalsy(); + + await cleanupTestDb(store); + }); }); // ============================================================================= From 5303f5049b9c6ba9a4191cbdf12a8042e3ab9f9a Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 24 Sep 2026 21:55:55 -0500 Subject: [PATCH 39/82] test(migrations): cover a legacy-table repair that fails on a stamped store When an older build recreates the legacy vector table on an index already at version 2, the next open repairs it; if that repair fails (here a partitioned table with different dimensions), the store must still open with a warning and leave the legacy table for a later attempt. The new test forces that failure and checks the warning, the version, and the layout. (cherry picked from commit a7d9e172fa9dcf970b9ce13666674b9cd1ec494f) --- test/store-migrations.test.ts | 19 ++++++++++++++++++- 1 file changed, 18 insertions(+), 1 deletion(-) diff --git a/test/store-migrations.test.ts b/test/store-migrations.test.ts index 466e6b69e..8bc567000 100644 --- a/test/store-migrations.test.ts +++ b/test/store-migrations.test.ts @@ -3,7 +3,7 @@ * per-collection partitioned layout, chunk by chunk, resumably, and drops the * legacy table in the transaction that stamps the version. */ -import { describe, test, expect, afterEach } from "vitest"; +import { describe, test, expect, afterEach, vi } from "vitest"; import { mkdtemp, rm } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; @@ -432,6 +432,23 @@ describe("runStoreMigrations", () => { expect(partitionCount(s.db, "b")).toBe(STANDARD_B_ROWS); }); + test("keeps the legacy table in place when the repair fails on a stamped store", async () => { + const s = await openStore(); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + seedStandardFixture(s.db); + s.db.exec(`PRAGMA user_version = ${VECTOR_PARTITION_VERSION}`); + createPartitionedVecTable(s.db, DIMS + 1); + expect(vecLayout(s.db).kind).toBe("legacy"); + const warn = vi.spyOn(console, "warn").mockImplementation(() => {}); + + expect(() => runStoreMigrations(s.db, { installFtsSyncTriggers: () => {}, sqliteVecAvailable: true })).not.toThrow(); + + expect(warn).toHaveBeenCalledWith(expect.stringContaining("Legacy vector table left in place")); + expect(getUserVersion(s.db)).toBe(VECTOR_PARTITION_VERSION); + expect(vecLayout(s.db).kind).toBe("legacy"); + warn.mockRestore(); + }); + test("leaves a stamped store without a legacy table alone", async () => { const s = await openStore(); createPartitionedVecTable(s.db, DIMS); From 58300dac73b2b9973ff0acc42683681703ebf551 Mon Sep 17 00:00:00 2001 From: Brett Date: Thu, 3 Sep 2026 10:24:19 -0500 Subject: [PATCH 40/82] feat(cleanup): repack the vector table when its chunks are mostly holes `qmd cleanup` compacts `vectors_vec` in place when fewer than 90% of its chunk reads hold live rows. vec0 puts each insert in the first free slot of its newest chunk and reclaims a chunk only once it is completely empty, so a delete in any older chunk leaves a hole that every brute-force scan still reads: re-embeds (DELETE then INSERT per chunk) and orphan removal accumulate them. A 794k-vector index measured at 36% occupancy (2,166 chunks for 776 needed) scanned in 1.8s where a packed copy of the same rows took 0.9s, and VACUUM never touches vec0's fixed-size chunk blobs. `vectorTableLayout` reads the chunk and row counts from vec0's shadow tables and reports the chunks a packed table would need. `repackVectors` moves the live rows of every chunk filled below 90% to the tail, deleting and re-inserting them one chunk per IMMEDIATE transaction, so the emptied chunks vanish, the write lock is never held for more than one chunk of rows, and an interrupted run leaves a consistent table that the next cleanup finishes. A legacy table without a `hash_seq` key is left for `ensureVecTable` to rebuild. The repack runs after orphan removal and before VACUUM; `previewCleanup` projects the layout after orphan removal so `--dry-run` predicts the same decision, and the cleanup output reports the chunk counts either way. Tests cover the layout arithmetic, the repack keeping every live row and the nearest-neighbour answer, a packed table left alone, the preview not rebuilding, a failing chunk move rolling back that chunk, the preview matching the run when orphans are pending, a legacy table, and a database without a vector table. (cherry picked from commit 5ddf8ed44e8cc7d2a77c0eb2bd96be00a660b233) --- CHANGELOG.md | 12 ++++ README.md | 2 +- src/cli/qmd.ts | 19 +++++- src/maintenance.ts | 18 +++++- src/store.ts | 139 +++++++++++++++++++++++++++++++++++++++++-- test/cleanup.test.ts | 139 +++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 320 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8ae49a426..ed15f73b3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -114,6 +114,18 @@ to an older qmd afterwards needs `qmd embed -f`, and a `qmd mcp` server started before the upgrade must be restarted. #983 (thanks @brettdavies) +### Changed + +- `qmd cleanup` repacks the vector table when fewer than 90% of its chunk + reads hold live rows. vec0 fills the first free slot of its newest chunk + and reclaims a chunk only once it is empty, so re-embeds and orphan removal + leave holes in older chunks that every brute-force scan still reads; a + 794k-vector index at 36% occupancy scanned twice as slowly as a packed copy + of the same rows. The repack moves the live rows of each mostly-empty chunk + to the tail, one chunk per short transaction, so it never holds the write + lock for long and an interrupted run leaves a consistent table. The cleanup + output and `qmd cleanup --dry-run` report the chunk counts. + ## [2.8.3] - 2026-08-16 ### Security diff --git a/README.md b/README.md index d6ee6ed0e..b89af6960 100644 --- a/README.md +++ b/README.md @@ -1335,7 +1335,7 @@ qmd multi-get "docs/*.md" --max-bytes 20480 # Output multi-get as JSON for agent processing qmd multi-get "docs/*.md" --json -# Clean up cache and orphaned data +# Drop caches and orphans; repack the vector table when it has holes qmd cleanup ``` diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index e6e834ebb..6978380f7 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -62,6 +62,7 @@ import { countOrphanedVectors, previewCleanup, runCleanup, + type VectorTableLayout, getCollectionsWithoutContext, getTopLevelPathsWithoutContext, handelize, @@ -2906,6 +2907,10 @@ function outputResults(results: OutputRow[], query: string, opts: OutputOptions) warnUnresolvedFullPaths(unresolvedCount, filtered.length); } +function describeVectorLayout(layout: VectorTableLayout): string { + return `${layout.chunks} chunks for ${layout.neededChunks} needed, ${Math.round(layout.occupancy * 100)}% useful`; +} + // Resolve -c collection filter: supports single string, array, or undefined. // Returns validated collection names (exits on unknown collection). function resolveCollectionFilter(raw: string | string[] | undefined, useDefaults: boolean = false): string[] { @@ -5246,12 +5251,19 @@ if (isMain) { if (stats.orphanedContent > 0) { console.log(`Would remove ${stats.orphanedContent} orphaned content hashes`); } + if (stats.vectorLayout) { + console.log(stats.vectorsRepacked + ? `Would repack the vector table (${describeVectorLayout(stats.vectorLayout)})` + : `${c.dim}Vector table is packed (${describeVectorLayout(stats.vectorLayout)})${c.reset}`); + } console.log("Would compact FTS and vacuum the database"); closeDb(); break; } - const stats = runCleanup(db); + const stats = runCleanup(db, { + onVectorRepack: (layout) => console.log(`Repacking the vector table (${describeVectorLayout(layout)})...`), + }); console.log(`${c.green}✓${c.reset} Cleared ${stats.cacheCount} cached API responses`); if (stats.orphanedVectors > 0) { console.log(`${c.green}✓${c.reset} Removed ${stats.orphanedVectors} orphaned embedding chunks`); @@ -5264,6 +5276,11 @@ if (isMain) { if (stats.orphanedContent > 0) { console.log(`${c.green}✓${c.reset} Removed ${stats.orphanedContent} orphaned content hashes`); } + if (stats.vectorLayout) { + console.log(stats.vectorsRepacked + ? `${c.green}✓${c.reset} Repacked the vector table (${describeVectorLayout(stats.vectorLayout)})` + : `${c.dim}Vector table is packed (${describeVectorLayout(stats.vectorLayout)})${c.reset}`); + } console.log(`${c.green}✓${c.reset} FTS compacted, database vacuumed`); closeDb(); diff --git a/src/maintenance.ts b/src/maintenance.ts index 5393555b3..9af91b431 100644 --- a/src/maintenance.ts +++ b/src/maintenance.ts @@ -15,8 +15,11 @@ import { clearAllEmbeddings, optimizeDocumentsFts, previewCleanup, + repackVectors, runCleanup, + vectorTableLayout, type CleanupStats, + type VectorTableLayout, } from "./store.js"; export class Maintenance { @@ -61,14 +64,25 @@ export class Maintenance { optimizeDocumentsFts(this.store.db); } + /** Chunk layout of the vector table, or null without one. */ + vectorLayout(): VectorTableLayout | null { + return vectorTableLayout(this.store.db); + } + + /** Move the live rows of mostly-empty vector chunks to the tail; see {@link repackVectors}. */ + repackVectors(): VectorTableLayout | null { + return repackVectors(this.store.db); + } + /** Preview what {@link run} would remove, without writing. */ preview(): CleanupStats { return previewCleanup(this.store.db); } /** - * Full cleanup: cache, orphaned vectors, inactive docs, orphaned content, - * FTS optimize, vacuum. Same sequence as `qmd cleanup`. + * Full cleanup: cache, orphaned vectors, vector repack when the table is + * mostly holes, inactive docs, orphaned content, FTS optimize, vacuum. Same + * sequence as `qmd cleanup`. */ run(): CleanupStats { return runCleanup(this.store.db); diff --git a/src/store.ts b/src/store.ts index c898d4df3..6d1149f58 100644 --- a/src/store.ts +++ b/src/store.ts @@ -3259,6 +3259,117 @@ export function cleanupOrphanedVectors(db: Database): number { }); } +/** + * Repack the vector table when fewer than this share of its chunk reads hold + * live rows. vec0 places an insert in the first free slot of its newest chunk + * and reclaims a chunk only once it is empty, so a delete in any older chunk + * leaves a hole that every brute-force scan still reads; at 36% occupancy a + * scan took twice as long as on a packed copy of the same rows. + */ +const VEC_REPACK_BELOW_OCCUPANCY = 0.9; + +/** Chunks filled below this share have their live rows moved to the tail. */ +const VEC_REPACK_CHUNK_FILL = 0.9; + +export type VectorTableLayout = { + rows: number; + chunks: number; + /** Chunks a freshly packed table needs for the same rows. */ + neededChunks: number; + /** neededChunks over chunks: the share of a scan's chunk reads that hold live rows. */ + occupancy: number; +}; + +interface VecChunkRow { + chunkId: number; + size: number; + validity: Uint8Array; + rowids: Uint8Array; +} + +function vecTableReadable(db: Database): boolean { + if (!isSqliteVecAvailable()) return false; + try { + db.prepare(`SELECT 1 FROM vectors_vec LIMIT 0`).get(); + return true; + } catch { + return false; + } +} + +/** + * Chunk layout of vectors_vec from vec0's shadow tables, or null without a + * readable table. `droppedRows` projects the layout after that many rows are + * deleted without their chunks going away, which is what orphan removal does. + */ +export function vectorTableLayout(db: Database, droppedRows: number = 0): VectorTableLayout | null { + if (!vecTableReadable(db)) return null; + const chunkRow = db.prepare(`SELECT COUNT(*) AS chunks, MAX(size) AS chunkSize FROM vectors_vec_chunks`).get() as { chunks: number; chunkSize: number | null }; + const stored = (db.prepare(`SELECT COUNT(*) AS c FROM vectors_vec_rowids`).get() as { c: number }).c; + const rows = Math.max(stored - droppedRows, 0); + const chunkSize = chunkRow.chunkSize ?? 0; + const neededChunks = chunkSize > 0 ? Math.ceil(rows / chunkSize) : 0; + const occupancy = chunkRow.chunks > 0 ? neededChunks / chunkRow.chunks : 1; + return { rows, chunks: chunkRow.chunks, neededChunks, occupancy }; +} + +/** Rows of vectors_vec that cleanupOrphanedVectors would delete: orphaned chunks that still hold a vector. */ +function countOrphanedVectorRows(db: Database): number { + if (!vecTableReadable(db)) return 0; + return withLazyContentVectorMigration(db, () => (db.prepare(` + SELECT COUNT(*) AS c + FROM content_vectors cv + WHERE NOT EXISTS (SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1) + AND EXISTS (SELECT 1 FROM vectors_vec_rowids r WHERE r.id = cv.hash || '_' || cv.seq) + `).get() as { c: number }).c); +} + +function liveSlots(chunk: VecChunkRow): number[] { + const slots: number[] = []; + for (let i = 0; i < chunk.size; i++) { + if ((chunk.validity[i >> 3]! >> (i & 7)) & 1) slots.push(i); + } + return slots; +} + +/** + * Compact vectors_vec in place: the live rows of every chunk that is mostly + * holes are deleted and re-inserted, one chunk per IMMEDIATE transaction. + * vec0 puts each insert in the first free slot of its newest chunk and drops + * a chunk once it is empty, so the moved rows fill the tail and the emptied + * chunks vanish. A transaction holds the write lock for at most one chunk of + * rows, so concurrent embed and update runs interleave with it, and an + * interrupted run leaves a consistent table that the next cleanup finishes. + * A legacy table without a hash_seq key is left for ensureVecTable to rebuild. + */ +export function repackVectors(db: Database, onChunk?: (moved: number, total: number) => void): VectorTableLayout | null { + if (!vecTableReadable(db)) return null; + const ddl = (db.prepare(`SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'vectors_vec'`).get() as { sql: string }).sql; + if (!ddl.includes("hash_seq")) return vectorTableLayout(db); + const chunks = db.prepare(`SELECT chunk_id AS chunkId, size, validity, rowids FROM vectors_vec_chunks ORDER BY chunk_id`).all() as VecChunkRow[]; + const newest = chunks.length > 0 ? chunks[chunks.length - 1]!.chunkId : -1; + const sparse = chunks.filter((c) => c.chunkId !== newest && liveSlots(c).length < c.size * VEC_REPACK_CHUNK_FILL); + const keyOf = db.prepare(`SELECT id FROM vectors_vec_rowids WHERE rowid = ?`); + const vectorOf = db.prepare(`SELECT embedding FROM vectors_vec WHERE hash_seq = ?`); + const remove = db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`); + const insert = db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`); + sparse.forEach((chunk, i) => { + const rowids = new DataView(chunk.rowids.buffer, chunk.rowids.byteOffset, chunk.rowids.byteLength); + db.transaction(() => { + for (const slot of liveSlots(chunk)) { + const key = (keyOf.get(rowids.getBigInt64(slot * 8, true)) as { id: string } | undefined)?.id; + if (key === undefined) continue; + const row = vectorOf.get(key) as { embedding: Uint8Array } | undefined; + if (row === undefined) continue; + remove.run(key); + insert.run(key, row.embedding); + } + }).immediate(); + onChunk?.(i + 1, sparse.length); + }); + return vectorTableLayout(db); +} + /** * Run VACUUM to reclaim unused space in the database. * This operation rebuilds the database file to eliminate fragmentation. @@ -3286,6 +3397,10 @@ export type CleanupStats = { orphanedVectors: number; inactiveDocs: number; orphanedContent: number; + /** Chunk layout of vectors_vec before any repack, null without a vector table. */ + vectorLayout: VectorTableLayout | null; + /** True when the vector table was repacked (or, in a preview, would be). */ + vectorsRepacked: boolean; }; /** Counts what `runCleanup` would remove, including content only held by inactive docs. */ @@ -3294,21 +3409,35 @@ export function previewCleanup(db: Database): CleanupStats { const orphanedVectors = countOrphanedVectors(db); const inactiveDocs = (db.prepare(`SELECT COUNT(*) as c FROM documents WHERE active = 0`).get() as { c: number }).c; const orphanedContent = countOrphanedContent(db); - return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent }; + const vectorLayout = vectorTableLayout(db, countOrphanedVectorRows(db)); + const vectorsRepacked = vectorLayout !== null && vectorLayout.occupancy < VEC_REPACK_BELOW_OCCUPANCY; + return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent, vectorLayout, vectorsRepacked }; } +export type CleanupHooks = { + /** Called with the layout right before a vector repack starts. */ + onVectorRepack?: (layout: VectorTableLayout) => void; +}; + /** - * Full `qmd cleanup` sequence: drop cache, orphaned vectors, inactive document - * rows, then the content those rows were pinning, compact FTS5, vacuum. + * Full `qmd cleanup` sequence: drop cache, orphaned vectors, repack the vector + * table when its chunks are mostly holes, inactive document rows, then the + * content those rows were pinning, compact FTS5, vacuum. */ -export function runCleanup(db: Database): CleanupStats { +export function runCleanup(db: Database, hooks: CleanupHooks = {}): CleanupStats { const cacheCount = deleteLLMCache(db); const orphanedVectors = cleanupOrphanedVectors(db); + const vectorLayout = vectorTableLayout(db); + const vectorsRepacked = vectorLayout !== null && vectorLayout.occupancy < VEC_REPACK_BELOW_OCCUPANCY; + if (vectorsRepacked && vectorLayout) { + hooks.onVectorRepack?.(vectorLayout); + repackVectors(db); + } const inactiveDocs = deleteInactiveDocuments(db); const orphanedContent = cleanupOrphanedContent(db); optimizeDocumentsFts(db); vacuumDatabase(db); - return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent }; + return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent, vectorLayout, vectorsRepacked }; } // ============================================================================= diff --git a/test/cleanup.test.ts b/test/cleanup.test.ts index c9ec9c02f..0bde3756e 100644 --- a/test/cleanup.test.ts +++ b/test/cleanup.test.ts @@ -16,9 +16,12 @@ import { cleanupOrphanedContent, countOrphanedContent, previewCleanup, + repackVectors, runCleanup, + vectorTableLayout, type Store, } from "../src/store.js"; +import type { Database } from "../src/db.js"; let store: Store | null = null; @@ -107,3 +110,139 @@ describe("qmd cleanup reclaim (#550)", () => { expect(s.db.prepare(`SELECT count(*) as c FROM documents`).get()).toEqual({ c: 1 }); }); }); + +describe("qmd cleanup vector repack", () => { + const DIMS = 3; + + /** + * Inserts `total` vectors, then deletes every one except the indexes in + * `keep` (which also get active documents). vec0 drops a chunk only when it + * is completely empty, so keeping one row in each chunk leaves the holes. + */ + async function seedVectors(s: Store, total: number, keep: readonly number[]): Promise { + const now = new Date().toISOString(); + s.ensureVecTable(DIMS); + s.db.transaction(() => { + for (let i = 0; i < total; i++) { + const hash = `vec${String(i).padStart(5, "0")}`; + if (keep.includes(i)) { + insertContent(s.db, hash, `body ${hash}`, now); + insertDocument(s.db, "docs", `${hash}.md`, hash, hash, now, now); + } + s.insertEmbedding(hash, 0, 0, new Float32Array([1, i * 0.001, 0]), "test-model", now, 1); + } + const drop = s.db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`); + for (let i = 0; i < total; i++) if (!keep.includes(i)) drop.run(`vec${String(i).padStart(5, "0")}_0`); + })(); + } + + /** Two chunks of 1024 slots holding three live rows: two in the first chunk, one in the second. */ + const HOLEY = { total: 1100, keep: [0, 1, 1099] } as const; + + function nearest(s: Store): string { + const row = s.db.prepare(`SELECT hash_seq FROM vectors_vec WHERE embedding MATCH ? AND k = 1`).get(new Float32Array([1, 0, 0])) as { hash_seq: string }; + return row.hash_seq; + } + + test("layout counts the chunks a packed table would need against the chunks in use", async () => { + const s = await openStore(); + await seedVectors(s, HOLEY.total, HOLEY.keep); + + expect(vectorTableLayout(s.db)).toEqual({ rows: 3, chunks: 2, neededChunks: 1, occupancy: 0.5 }); + }); + + test("layout is null without a vector table", async () => { + const s = await openStore(); + expect(vectorTableLayout(s.db)).toBeNull(); + expect(previewCleanup(s.db)).toMatchObject({ vectorLayout: null, vectorsRepacked: false }); + }); + + test("runCleanup repacks a table that is mostly holes and keeps every live row", async () => { + const s = await openStore(); + await seedVectors(s, HOLEY.total, HOLEY.keep); + + const stats = runCleanup(s.db); + expect(stats.vectorsRepacked).toBe(true); + expect(stats.vectorLayout).toMatchObject({ chunks: 2, neededChunks: 1 }); + expect(vectorTableLayout(s.db)).toEqual({ rows: 3, chunks: 1, neededChunks: 1, occupancy: 1 }); + expect(s.db.prepare(`SELECT hash_seq FROM vectors_vec ORDER BY hash_seq`).all()).toEqual([ + { hash_seq: "vec00000_0" }, { hash_seq: "vec00001_0" }, { hash_seq: "vec01099_0" }, + ]); + expect(nearest(s)).toBe("vec00000_0"); + }); + + test("runCleanup leaves a packed table alone", async () => { + const s = await openStore(); + await seedVectors(s, 3, [0, 1, 2]); + + const stats = runCleanup(s.db); + expect(stats.vectorsRepacked).toBe(false); + expect(stats.vectorLayout).toEqual({ rows: 3, chunks: 1, neededChunks: 1, occupancy: 1 }); + }); + + test("previewCleanup reports the repack without rebuilding", async () => { + const s = await openStore(); + await seedVectors(s, HOLEY.total, HOLEY.keep); + + expect(previewCleanup(s.db)).toMatchObject({ vectorsRepacked: true, vectorLayout: { chunks: 2, neededChunks: 1 } }); + expect(vectorTableLayout(s.db)).toMatchObject({ chunks: 2 }); + }); + + test("a chunk move that fails midway leaves that chunk's rows in place", async () => { + const s = await openStore(); + await seedVectors(s, HOLEY.total, HOLEY.keep); + const failing: Database = { + prepare: (sql: string) => { + const real = s.db.prepare(sql); + if (!sql.startsWith("INSERT INTO vectors_vec (")) return real; + return { ...real, get: real.get.bind(real), all: real.all.bind(real), iterate: real.iterate.bind(real), run: () => { throw new Error("injected failure during re-insert"); } }; + }, + transaction: (fn) => s.db.transaction(fn), + exec: (sql: string) => s.db.exec(sql), + loadExtension: (path: string) => s.db.loadExtension(path), + close: () => s.db.close(), + }; + + expect(() => repackVectors(failing)).toThrow("injected failure during re-insert"); + expect(vectorTableLayout(s.db)).toEqual({ rows: 3, chunks: 2, neededChunks: 1, occupancy: 0.5 }); + expect(nearest(s)).toBe("vec00000_0"); + expect(s.db.prepare(`SELECT hash_seq FROM vectors_vec ORDER BY hash_seq`).all()).toEqual([ + { hash_seq: "vec00000_0" }, { hash_seq: "vec00001_0" }, { hash_seq: "vec01099_0" }, + ]); + }); + + test("previewCleanup projects the layout after orphan removal, matching what runCleanup does", async () => { + const s = await openStore(); + const now = new Date().toISOString(); + s.ensureVecTable(DIMS); + s.db.transaction(() => { + for (let i = 0; i < 2048; i++) { + const hash = `orphan${String(i).padStart(5, "0")}`; + insertContent(s.db, hash, `body ${hash}`, now); + insertDocument(s.db, "docs", `${hash}.md`, hash, hash, now, now); + s.insertEmbedding(hash, 0, 0, new Float32Array([1, i * 0.001, 0]), "test-model", now, 1); + } + })(); + for (let i = 0; i < 2048; i += 2) deactivateDocument(s.db, "docs", `orphan${String(i).padStart(5, "0")}.md`); + + expect(vectorTableLayout(s.db)).toEqual({ rows: 2048, chunks: 2, neededChunks: 2, occupancy: 1 }); + expect(previewCleanup(s.db)).toMatchObject({ orphanedVectors: 1024, vectorsRepacked: true, vectorLayout: { rows: 1024, chunks: 2, neededChunks: 1 } }); + + const stats = runCleanup(s.db); + expect(stats.vectorsRepacked).toBe(true); + expect(vectorTableLayout(s.db)).toEqual({ rows: 1024, chunks: 1, neededChunks: 1, occupancy: 1 }); + }); + + test("a legacy vector table without hash_seq is left alone", async () => { + const s = await openStore(); + s.ensureVecTable(DIMS); + s.db.exec(`DROP TABLE vectors_vec`); + s.db.exec(`CREATE VIRTUAL TABLE vectors_vec USING vec0(hash TEXT PRIMARY KEY, embedding float[${DIMS}] distance_metric=cosine)`); + s.db.transaction(() => { + for (let i = 0; i < 1100; i++) s.db.prepare(`INSERT INTO vectors_vec (hash, embedding) VALUES (?, ?)`).run(`legacy${i}`, new Float32Array([1, i * 0.001, 0])); + for (let i = 2; i < 1099; i++) s.db.prepare(`DELETE FROM vectors_vec WHERE hash = ?`).run(`legacy${i}`); + })(); + + expect(repackVectors(s.db)).toEqual({ rows: 3, chunks: 2, neededChunks: 1, occupancy: 0.5 }); + }); +}); From 8e7fb489f4449fe0aeffaa8f306c9ad2f70a5f4e Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:41:04 -0500 Subject: [PATCH 41/82] test(doctor): bound the whole vector sample check, not just its query #978's worker bounds getEmbeddingVectorSamples. This worker runs the doctor's checkEmbeddingVectorSamples end to end (sample, read bodies, chunk, re-embed, compare) over 16 documents of 64 KiB, each with 8 active paths and 32 stored chunks, under a 32 MiB SQLite heap limit, and expects it to report ok. The heap limit binds under bun only: better-sqlite3 builds SQLite with DEFAULT_MEMSTATUS=0, so Node accepts the pragma without enforcing it. --- test/_helpers/doctor-vector-check-worker.ts | 36 +++++++++++++++++++++ test/doctor-vector-sampling.test.ts | 11 +++++++ 2 files changed, 47 insertions(+) create mode 100644 test/_helpers/doctor-vector-check-worker.ts diff --git a/test/_helpers/doctor-vector-check-worker.ts b/test/_helpers/doctor-vector-check-worker.ts new file mode 100644 index 000000000..18a7e6be1 --- /dev/null +++ b/test/_helpers/doctor-vector-check-worker.ts @@ -0,0 +1,36 @@ +import { createStore } from "../../src/store.js"; +import { checkEmbeddingVectorSamples } from "../../src/cli/qmd.js"; +import { LlamaCpp, setDefaultLlamaCpp } from "../../src/llm.js"; + +// Every stored vector is [1, 0] and every re-embedded chunk returns [1, 0], so +// the check passes whenever it can sample, read and chunk the bodies. +class ConstantLlm extends LlamaCpp { + async tokenize(text: string) { return new Array(Math.ceil(text.length / 16)).fill(1); } + async embed() { return { embedding: [1, 0], model: "model" }; } +} + +const store = createStore(":memory:"); +try { + setDefaultLlamaCpp(new ConstantLlm()); + const body = "word ".repeat(13_107); // About 64 KiB per document. + store.ensureVecTable(2); + store.db.transaction(() => { + for (let doc = 0; doc < 16; doc++) { + const hash = `document-${doc}`; + store.insertContent(hash, body, "2026-01-01"); + for (let path = 0; path < 8; path++) { + store.insertDocument("test", `${hash}-${path}.md`, hash, hash, "2026-01-01", "2026-01-01"); + } + for (let seq = 0; seq < 32; seq++) { + store.insertEmbedding(hash, seq, 0, new Float32Array([1, 0]), "model", "2026-01-01", 32, "current"); + } + } + })(); + store.db.exec("PRAGMA temp_store = MEMORY"); + store.db.exec("PRAGMA hard_heap_limit = 33554432"); + const result = await checkEmbeddingVectorSamples(store.db, "model", "current"); + console.log(JSON.stringify(result)); +} finally { + setDefaultLlamaCpp(null); + store.close(); +} diff --git a/test/doctor-vector-sampling.test.ts b/test/doctor-vector-sampling.test.ts index b563b98fa..faff96c8a 100644 --- a/test/doctor-vector-sampling.test.ts +++ b/test/doctor-vector-sampling.test.ts @@ -109,4 +109,15 @@ describe("doctor vector sampling", () => { expect(result.status).toBe(0); expect(JSON.parse(result.stdout)).toEqual({ samples: 3, distinctChunks: 3, bodiesComplete: true }); }); + + test("the doctor check verifies large documents with duplicate paths within a 32 MiB SQLite budget", () => { + // SQLite's hard heap limit is process-wide and cannot be raised again. + const worker = join(projectRoot, "test", "_helpers", "doctor-vector-check-worker.ts"); + const args = isBun ? [worker] : [join(projectRoot, "node_modules", "tsx", "dist", "cli.mjs"), worker]; + const result = spawnSync(process.execPath, args, { encoding: "utf8", timeout: 60_000 }); + expect(result.error).toBeUndefined(); + expect(result.stderr).toBe(""); + expect(result.status).toBe(0); + expect(JSON.parse(result.stdout)).toEqual({ ok: true, details: "3 sampled chunks reproduce stored vectors" }); + }); }); From 6af0bc7bc66b54e31f330aa366f8871acc6950ce Mon Sep 17 00:00:00 2001 From: Michael Averto Date: Sat, 26 Sep 2026 11:47:42 -0400 Subject: [PATCH 42/82] fix(store): load legacy adoption sample body by rowid instead of per-chunk GROUP BY maybeAdoptLegacyEmbeddingFingerprint joined content and grouped by c.doc, so SQLite materialized the full document body once per legacy chunk (and per active path) before LIMIT 1 discarded it. The cost grows with chunks x active paths x body size; the same pattern exists in the doctor vector-sample check (#978). Pick the sample row through indexes first (EXISTS filters for an active document and for content, same ORDER BY hash, seq), then load only that row's body and first active path by rowid. Both reads share one transaction. The sampled chunk and everything downstream are unchanged. Refs #994 (cherry picked from commit 0af7eab2a6fad5ad581f3380c80e165c8e2c8ee3) --- CHANGELOG.md | 7 + src/store.ts | 43 +++-- .../_helpers/legacy-adoption-sample-worker.ts | 46 ++++++ test/store.test.ts | 150 +++++++++++++++++- 4 files changed, 235 insertions(+), 11 deletions(-) create mode 100644 test/_helpers/legacy-adoption-sample-worker.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 9aae599e4..4440b1343 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,6 +22,13 @@ with the store-selected embedding model instead of the global default. This keeps chunk boundaries aligned with the model that creates and verifies the stored vectors without initializing an unrelated provider. +- `qmd doctor` no longer stalls or fills the temp directory on large indexes + when checking legacy (empty-fingerprint) embeddings. The adoption sample + query joined `content` and grouped by the document body, so SQLite + materialized the full body once per legacy chunk and per active path before + `LIMIT 1` discarded it, the same pattern as the doctor vector-sample check + (#978). It now picks the sample row through indexes and loads only that + row's body; the sampled chunk is unchanged. #994 (thanks @mjaverto) ### Changed diff --git a/src/store.ts b/src/store.ts index ff5f76677..cf532d7c8 100644 --- a/src/store.ts +++ b/src/store.ts @@ -2639,16 +2639,39 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store: Store, model: return { checked: false, adopted: 0, reason: "no legacy empty-fingerprint embeddings" }; } - const sample = withLazyContentVectorMigration(db, () => db.prepare(` - SELECT cv.hash, cv.seq, cv.pos, cv.total_chunks, c.doc AS body, MIN(d.path) AS path - FROM content_vectors cv - JOIN documents d ON d.hash = cv.hash AND d.active = 1 - JOIN content c ON c.hash = cv.hash - WHERE cv.model = ? AND cv.embed_fingerprint = '' - GROUP BY cv.hash, cv.seq, cv.pos, cv.total_chunks, c.doc - ORDER BY cv.hash, cv.seq - LIMIT 1 - `).get(model) as { hash: string; seq: number; pos: number; total_chunks: number; body: string; path: string } | undefined); + // Pick the sample through indexes, then read one body by rowid; a join that + // carries c.doc costs one body copy per legacy chunk × active path. Both + // EXISTS filters depend only on hash, so this is still the lowest (hash, seq). + // One read transaction keeps the pick and the load on the same snapshot. + const sample = withLazyContentVectorMigration(db, () => { + const pickSample = db.prepare(` + SELECT cv.rowid AS rid + FROM content_vectors cv + WHERE cv.model = ? AND cv.embed_fingerprint = '' + AND cv.hash = ( + SELECT l.hash + FROM content_vectors l + WHERE l.model = ? AND l.embed_fingerprint = '' + AND EXISTS (SELECT 1 FROM documents d WHERE d.hash = l.hash AND d.active = 1) + AND EXISTS (SELECT 1 FROM content c WHERE c.hash = l.hash) + ORDER BY l.hash + LIMIT 1 + ) + ORDER BY cv.seq + LIMIT 1 + `); + const loadSample = db.prepare(` + SELECT cv.hash, cv.seq, cv.pos, cv.total_chunks, c.doc AS body, + (SELECT MIN(d.path) FROM documents d WHERE d.hash = cv.hash AND d.active = 1) AS path + FROM content_vectors cv + JOIN content c ON c.hash = cv.hash + WHERE cv.rowid = ? + `); + return db.transaction(() => { + const pick = pickSample.get(model, model) as { rid: number } | null | undefined; + return pick ? loadSample.get(pick.rid) : undefined; + })() as { hash: string; seq: number; pos: number; total_chunks: number; body: string; path: string } | null | undefined; + }); if (!sample) { return { checked: false, adopted: 0, reason: `${legacyCount} legacy docs have no active sample` }; diff --git a/test/_helpers/legacy-adoption-sample-worker.ts b/test/_helpers/legacy-adoption-sample-worker.ts new file mode 100644 index 000000000..768da69dd --- /dev/null +++ b/test/_helpers/legacy-adoption-sample-worker.ts @@ -0,0 +1,46 @@ +/** + * legacy-adoption-sample-worker - runs legacy fingerprint adoption under a + * 128 MiB SQLite heap limit. + * + * Spawned by store.test.ts: PRAGMA hard_heap_limit is process-wide and cannot + * be raised again, so it must not leak into the test runner. The fixture has + * large bodies with many legacy chunks and duplicate paths; materializing one + * body per chunk (per path) needs far more than the budget. + */ +import { createStore, maybeAdoptLegacyEmbeddingFingerprint } from "../../src/store.ts"; + +const model = "model"; +const store = createStore(":memory:"); +try { + const body = "word ".repeat(13_107); // About 64 KiB per document. + const insertChunk = store.db.prepare(` + INSERT INTO content_vectors (hash, seq, model, embed_fingerprint, embedded_at) + VALUES (?, ?, ?, '', '2026-01-01') + `); + store.db.transaction(() => { + for (let doc = 0; doc < 16; doc++) { + const hash = `document-${doc}`; + store.insertContent(hash, body, "2026-01-01"); + for (let path = 0; path < 8; path++) { + store.insertDocument("test", `${hash}-${path}.md`, hash, hash, "2026-01-01", "2026-01-01"); + } + for (let seq = 0; seq < 32; seq++) insertChunk.run(hash, seq, model); + } + })(); + // Store the sample chunk's vector: the one the stub embedder returns. + store.ensureVecTable(3); + store.insertEmbedding("document-0", 0, 0, new Float32Array([0.1, 0.2, 0.3]), model, "2026-01-01", 32, ""); + // Adoption only tokenizes, detokenizes and embeds; a stub covers that surface. + store.llm = { + async tokenize(text: string) { return new Array(Math.max(1, Math.ceil(text.length / 16))).fill(1); }, + async detokenize(tokens: readonly number[]) { return "x".repeat(tokens.length * 16); }, + async embed() { return { embedding: [0.1, 0.2, 0.3], model }; }, + } as any; + + store.db.exec("PRAGMA temp_store = MEMORY"); + store.db.exec("PRAGMA hard_heap_limit = 134217728"); + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, model); + console.log(JSON.stringify(result)); +} finally { + store.close(); +} diff --git a/test/store.test.ts b/test/store.test.ts index 51f913a3d..13640d869 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -7,11 +7,13 @@ */ import { describe, test, expect, beforeAll, afterAll, beforeEach, afterEach, vi } from "vitest"; -import { openDatabase, loadSqliteVec } from "../src/db.js"; +import { openDatabase, loadSqliteVec, isBun } from "../src/db.js"; import type { Database } from "../src/db.js"; import { unlink, mkdtemp, rmdir, writeFile, rm, mkdir, rename, chmod, readFile, symlink } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; import YAML from "yaml"; import * as llmModule from "../src/llm.js"; import { disposeDefaultLlamaCpp, setDefaultLlamaCpp } from "../src/llm.js"; @@ -4191,6 +4193,152 @@ describe("Embedding batching", () => { } }); + const legacyModel = "hf:test/legacy-sample.gguf"; + // createFakeEmbedLlm().embed returns this vector for every text. + const matchingVector = new Float32Array([0.1, 0.2, 0.3]); + const otherVector = new Float32Array([0.3, -0.2, 0.1]); + + function addLegacyChunk( + store: Store, + hash: string, + seq: number, + opts: { model?: string; fingerprint?: string; vector?: Float32Array } = {}, + ): void { + const model = opts.model ?? legacyModel; + const fingerprint = opts.fingerprint ?? ""; + const embeddedAt = new Date(0).toISOString(); + if (opts.vector) { + store.insertEmbedding(hash, seq, 0, opts.vector, model, embeddedAt, 3, fingerprint); + } else { + store.db.prepare(` + INSERT INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at) + VALUES (?, ?, 0, ?, ?, 3, ?) + `).run(hash, seq, model, fingerprint, embeddedAt); + } + } + + test("legacy fingerprint adoption samples the lowest hash and seq when active documents share content", async () => { + const store = await createTestStore(); + const db = store.db; + const fakeLlm = createFakeEmbedLlm(); + store.llm = fakeLlm as any; + + try { + store.ensureVecTable(3); + await insertTestDocument(db, "docs", { hash: "hash-b", body: "Other legacy body.", displayPath: "other.md" }); + addLegacyChunk(store, "hash-b", 0, { vector: otherVector }); + // One body behind two active paths and an inactive one, with seqs inserted + // out of order so rowid order disagrees with ORDER BY hash, seq. + const body = "Shared legacy body without a heading."; + await insertTestDocument(db, "docs", { hash: "hash-a", body, displayPath: "z.md" }); + await insertTestDocument(db, "docs", { hash: "hash-a", body, displayPath: "b.md" }); + await insertTestDocument(db, "docs", { hash: "hash-a", body, displayPath: "a.md", active: 0 }); + addLegacyChunk(store, "hash-a", 2); + addLegacyChunk(store, "hash-a", 1); + addLegacyChunk(store, "hash-a", 0, { vector: matchingVector }); + + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, legacyModel); + + expect(result).toMatchObject({ checked: true, adopted: 4 }); + expect(result.reason).toMatch(/^sample hash-a_0 matched/); + // The title comes from the first active path (b.md), not the inactive a.md. + expect(fakeLlm.embedCalls.map(call => call.text)).toEqual([formatDocForEmbedding(body, "b", legacyModel)]); + } finally { + await cleanupTestDb(store); + } + }); + + test("legacy fingerprint adoption skips rows without an active document or content", async () => { + const store = await createTestStore(); + const db = store.db; + const fakeLlm = createFakeEmbedLlm(); + store.llm = fakeLlm as any; + + try { + store.ensureVecTable(3); + await insertTestDocument(db, "docs", { hash: "hash-a", body: "Inactive body.", displayPath: "inactive.md", active: 0 }); + addLegacyChunk(store, "hash-a", 0, { vector: otherVector }); + await insertTestDocument(db, "docs", { hash: "hash-b", body: "Missing body.", displayPath: "missing.md" }); + addLegacyChunk(store, "hash-b", 0, { vector: otherVector }); + // With foreign keys on, deleting content cascades to the document; turn + // them off so an active document is left pointing at missing content. + db.exec("PRAGMA foreign_keys = OFF"); + db.prepare(`DELETE FROM content WHERE hash = ?`).run("hash-b"); + db.exec("PRAGMA foreign_keys = ON"); + + const skipped = await maybeAdoptLegacyEmbeddingFingerprint(store, legacyModel); + + expect(skipped).toEqual({ checked: false, adopted: 0, reason: "2 legacy docs have no active sample" }); + expect(fakeLlm.embedCalls).toHaveLength(0); + + await insertTestDocument(db, "docs", { hash: "hash-c", body: "Active body.", displayPath: "active.md" }); + addLegacyChunk(store, "hash-c", 0, { vector: matchingVector }); + + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, legacyModel); + + expect(result).toMatchObject({ checked: true, adopted: 3 }); + expect(result.reason).toMatch(/^sample hash-c_0 matched/); + } finally { + await cleanupTestDb(store); + } + }); + + test("legacy fingerprint adoption ignores other models and fingerprinted rows", async () => { + const store = await createTestStore(); + const db = store.db; + const fakeLlm = createFakeEmbedLlm(); + store.llm = fakeLlm as any; + + try { + store.ensureVecTable(3); + await insertTestDocument(db, "docs", { hash: "hash-a", body: "Other model body.", displayPath: "other-model.md" }); + addLegacyChunk(store, "hash-a", 0, { model: "hf:test/other-model.gguf", vector: otherVector }); + await insertTestDocument(db, "docs", { hash: "hash-b", body: "Fingerprinted body.", displayPath: "fingerprinted.md" }); + addLegacyChunk(store, "hash-b", 0, { fingerprint: "abc123", vector: otherVector }); + await insertTestDocument(db, "docs", { hash: "hash-c", body: "Legacy body.", displayPath: "legacy.md" }); + addLegacyChunk(store, "hash-c", 0, { vector: matchingVector }); + + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, legacyModel); + + expect(result).toMatchObject({ checked: true, adopted: 1 }); + expect(result.reason).toMatch(/^sample hash-c_0 matched/); + expect(db.prepare(`SELECT hash, embed_fingerprint FROM content_vectors ORDER BY hash`).all()).toEqual([ + { hash: "hash-a", embed_fingerprint: "" }, + { hash: "hash-b", embed_fingerprint: "abc123" }, + { hash: "hash-c", embed_fingerprint: getEmbeddingFingerprint(legacyModel) }, + ]); + + // No legacy rows remain for this model: a second pass is a no-op. + const again = await maybeAdoptLegacyEmbeddingFingerprint(store, legacyModel); + + expect(again).toEqual({ checked: false, adopted: 0, reason: "no legacy empty-fingerprint embeddings" }); + expect(fakeLlm.embedCalls).toHaveLength(1); + } finally { + await cleanupTestDb(store); + } + }); + + test("legacy fingerprint adoption loads one sample body within a 128 MiB SQLite budget", () => { + // SQLite's hard heap limit is process-wide and cannot be raised again, so + // the fixture runs in a worker. The limit binds only under Bun with an + // SQLite that tracks memory (Bun's bundled SQLite on Linux, Homebrew SQLite + // on macOS), where the old per-chunk query fails with SQLITE_NOMEM. Apple's + // system libsqlite3 and better-sqlite3 are built with + // SQLITE_DEFAULT_MEMSTATUS=0, so there this only checks the adoption result. + const projectRoot = fileURLToPath(new URL("..", import.meta.url)); + const worker = join(projectRoot, "test", "_helpers", "legacy-adoption-sample-worker.ts"); + const args = isBun ? [worker] : [join(projectRoot, "node_modules", "tsx", "dist", "cli.mjs"), worker]; + const result = spawnSync(process.execPath, args, { encoding: "utf8", timeout: 20_000 }); + + expect(result.error).toBeUndefined(); + expect(result.stderr).toBe(""); + expect(result.status).toBe(0); + const adoption = JSON.parse(result.stdout) as { checked: boolean; adopted: number; reason: string }; + // 16 documents x 32 legacy chunks, all adopted after one sample matched. + expect(adoption).toMatchObject({ checked: true, adopted: 512 }); + expect(adoption.reason).toMatch(/^sample document-0_0 matched/); + }); + test("generateEmbeddings flushes batches when maxDocsPerBatch is reached", async () => { const store = await createTestStore(); const db = store.db; From 42fb497c26ffc079a44ea5738fe76ad887201259 Mon Sep 17 00:00:00 2001 From: Parker Rex Date: Sat, 26 Sep 2026 12:01:38 -0500 Subject: [PATCH 43/82] fix(query): drop duplicate query expansions before searching (#921) The expansion model can emit the same line more than once, and nothing removed the repeats. store.expandQuery only dropped entries equal to the original query, then cached the rest as-is, and cache reads returned rows unchanged. hybridQuery and vectorSearchQuery run one search per entry, so a cached row with 12 identical hyde lines embedded and scanned 12 times. Add uniqueExpansions(), an exact (type, query) filter, and apply it to fresh model output before caching and to both cache-read branches, so rows already bloated by older versions are cleaned as they are read. vec and hyde entries with the same text are kept; they route to different searches. Co-Authored-By: Claude Opus 5.5 (cherry picked from commit 309843bd88d5e7d45a1924d37af18002742cf2b8) --- CHANGELOG.md | 7 +++++++ src/store.ts | 23 +++++++++++++++++++---- test/store.test.ts | 26 ++++++++++++++++++++++++++ 3 files changed, 52 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2d87775e0..586afdb2a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -61,6 +61,13 @@ of 20,000 files takes about 5 s instead of about 3 minutes. #1021 (thanks @brettdavies) +- `qmd query` and `qmd vsearch` now drop repeated query expansions before + searching. The expansion model can repeat a line, and the cache kept every + copy: one reported query ran 23 expansions where 9 were distinct, and each + copy ran its own search. Cached expansions are deduplicated when read, so + existing caches need no rebuild. Repeated lex lines used to count more than + once in the rank fusion, so result order can shift slightly. (#921) + ## [2.8.3] - 2026-08-16 ### Security diff --git a/src/store.ts b/src/store.ts index ae7752153..43a0cef3b 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4929,6 +4929,21 @@ function removeIncompleteEmbeddings(db: Database, expectedChunksByHash: Map(); + return expansions.filter((e) => { + const key = `${e.type}\n${e.query}`; + if (seen.has(key)) return false; + seen.add(key); + return true; + }); +} + export async function expandQuery(query: string, model: string = DEFAULT_QUERY_MODEL, db: Database, llmOverride?: LlamaCpp): Promise { // Check cache first — stored as JSON preserving types. Intent is // deliberately absent from both the cache key and the generation call: @@ -4944,9 +4959,9 @@ export async function expandQuery(query: string, model: string = DEFAULT_QUERY_M const rows = parsed as Array>; // Migrate old cache format: { type, text } → { type, query } if (rows.length > 0 && typeof rows[0]?.query === "string") { - return rows.map((r) => ({ type: r.type as ExpandedQuery["type"], query: String(r.query) })); + return uniqueExpansions(rows.map((r) => ({ type: r.type as ExpandedQuery["type"], query: String(r.query) }))); } else if (rows.length > 0 && typeof rows[0]?.text === "string") { - return rows.map((r) => ({ type: r.type as ExpandedQuery["type"], query: String(r.text) })); + return uniqueExpansions(rows.map((r) => ({ type: r.type as ExpandedQuery["type"], query: String(r.text) }))); } } catch { // Old cache format (pre-typed, newline-separated text) — re-expand @@ -4959,9 +4974,9 @@ export async function expandQuery(query: string, model: string = DEFAULT_QUERY_M // Map Queryable[] → ExpandedQuery[] (same shape, decoupled from llm.ts internals). // Filter out entries that duplicate the original query text. - const expanded: ExpandedQuery[] = results + const expanded = uniqueExpansions(results .filter(r => r.text !== query) - .map(r => ({ type: r.type, query: r.text })); + .map(r => ({ type: r.type, query: r.text }))); if (expanded.length > 0) { setCachedResult(db, cacheKey, JSON.stringify(expanded)); diff --git a/test/store.test.ts b/test/store.test.ts index 10e5d5b54..2e2f7b4e6 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -1374,6 +1374,32 @@ describe("Query expansion cache (#818)", () => { await cleanupTestDb(store); } }); + + test("hybridQuery embeds each distinct cached expansion once (#921)", async () => { + const store = await createTestStore(); + const embedModel = "hf:ggml-org/embeddinggemma-300M-GGUF/embeddinggemma-300M-Q8_0.gguf"; + const embedBatchSpy = vi.fn(async (texts: string[]) => texts.map(() => ({ embedding: [1, 2, 3], model: embedModel }))); + store.db.exec(`CREATE TABLE vectors_vec (hash_seq TEXT PRIMARY KEY, embedding BLOB)`); + store.llm = { embedModelName: embedModel, embedBatch: embedBatchSpy } as any; + store.searchVec = vi.fn(async () => [] as SearchResult[]) as any; + try { + // The row from #921: one hyde string cached 12 times, one vec string twice. + const cached = [ + ...Array.from({ length: 12 }, () => ({ type: "hyde", query: "musubi reconstruction guide" })), + { type: "vec", query: "methods for musubi" }, + { type: "vec", query: "methods for musubi" }, + { type: "lex", query: "musubi reconstruction" }, + ]; + store.setCachedResult(getCacheKey("expandQuery", { query: "musubi", model: DEFAULT_QUERY_MODEL }), JSON.stringify(cached)); + + await hybridQuery(store, "musubi", { limit: 5, minScore: 0, skipRerank: true, intent: "x" }); + + expect(embedBatchSpy).toHaveBeenCalledTimes(1); + expect(embedBatchSpy.mock.calls[0]![0]).toHaveLength(3); + } finally { + await cleanupTestDb(store); + } + }); }); From f69e97ee79a5b7d4d657a0c835367b0a7f8ceb76 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:44:01 -0500 Subject: [PATCH 44/82] test(store): bound legacy adoption by active paths alone #995's worker grows chunks and duplicate paths together. This worker holds the chunk count at one and puts a 1 MiB body behind 200 active paths, then runs legacy fingerprint adoption under a 16 MiB SQLite heap limit and expects the sample to be checked and adopted. A query that carries the body through its join copies it once per path. The index lives in a temp file because the FTS table stores each path's copy of the body. As in #995's own worker test, the limit binds under bun only. --- .../legacy-adoption-shared-body-worker.ts | 43 +++++++++++++++++++ test/store.test.ts | 14 ++++++ 2 files changed, 57 insertions(+) create mode 100644 test/_helpers/legacy-adoption-shared-body-worker.ts diff --git a/test/_helpers/legacy-adoption-shared-body-worker.ts b/test/_helpers/legacy-adoption-shared-body-worker.ts new file mode 100644 index 000000000..3962a769d --- /dev/null +++ b/test/_helpers/legacy-adoption-shared-body-worker.ts @@ -0,0 +1,43 @@ +/** + * legacy-adoption-shared-body-worker - runs legacy fingerprint adoption for + * one legacy chunk whose 1 MiB body sits behind 200 active paths, under a + * 16 MiB SQLite heap limit. + * + * A sample query that carries c.doc through its join copies the body once per + * active path, far past the budget, while the chunk count stays at one. The + * index lives in a temp file: the FTS table stores every path's copy of the + * body, which an in-memory database would count against the heap limit. + */ +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createStore, maybeAdoptLegacyEmbeddingFingerprint } from "../../src/store.ts"; + +const model = "model"; +const dir = mkdtempSync(join(tmpdir(), "qmd-legacy-shared-body-")); +const store = createStore(join(dir, "index.sqlite")); +try { + const body = "word ".repeat(209_716); // About 1 MiB. + store.insertContent("shared", body, "2026-01-01"); + store.db.transaction(() => { + for (let path = 0; path < 200; path++) { + store.insertDocument("test", `shared-${path}.md`, "shared", "shared", "2026-01-01", "2026-01-01"); + } + })(); + store.ensureVecTable(3); + store.insertEmbedding("shared", 0, 0, new Float32Array([0.1, 0.2, 0.3]), model, "2026-01-01", 1, ""); + // Adoption only tokenizes, detokenizes and embeds; a stub covers that surface. + store.llm = { + async tokenize(text: string) { return new Array(Math.max(1, Math.ceil(text.length / 16))).fill(1); }, + async detokenize(tokens: readonly number[]) { return "x".repeat(tokens.length * 16); }, + async embed() { return { embedding: [0.1, 0.2, 0.3], model }; }, + } as any; + + store.db.exec("PRAGMA temp_store = MEMORY"); + store.db.exec("PRAGMA hard_heap_limit = 16777216"); + const result = await maybeAdoptLegacyEmbeddingFingerprint(store, model); + console.log(JSON.stringify({ checked: result.checked, adopted: result.adopted })); +} finally { + store.close(); + rmSync(dir, { recursive: true, force: true }); +} diff --git a/test/store.test.ts b/test/store.test.ts index 13640d869..a559c5724 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -4339,6 +4339,20 @@ describe("Embedding batching", () => { expect(adoption.reason).toMatch(/^sample document-0_0 matched/); }); + test("legacy fingerprint adoption reads one copy of a body shared by 200 active paths within a 16 MiB SQLite budget", () => { + // One legacy chunk, so only the active-path count multiplies the body. The + // limit binds under Bun only, as in the 128 MiB test above. + const projectRoot = fileURLToPath(new URL("..", import.meta.url)); + const worker = join(projectRoot, "test", "_helpers", "legacy-adoption-shared-body-worker.ts"); + const args = isBun ? [worker] : [join(projectRoot, "node_modules", "tsx", "dist", "cli.mjs"), worker]; + const result = spawnSync(process.execPath, args, { encoding: "utf8", timeout: 60_000 }); + + expect(result.error).toBeUndefined(); + expect(result.stderr).toBe(""); + expect(result.status).toBe(0); + expect(JSON.parse(result.stdout)).toEqual({ checked: true, adopted: 1 }); + }); + test("generateEmbeddings flushes batches when maxDocsPerBatch is reached", async () => { const store = await createTestStore(); const db = store.db; From 9908c6f9ae54ed0fbfe89bf502c09a74b207e7e6 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:51:21 -0500 Subject: [PATCH 45/82] test(store): cover #962's reindex fast path, size guard and body cap #962's three commits carry no tests. These pin what they change: - an unchanged mtime and size count the file as unchanged without a read: a same-size rewrite with its mtime put back stays unindexed, and a file made unreadable is still reported unchanged; - a file over 10 MB is skipped with FILE_TOO_LARGE and gets no document row; - a previously indexed file rewritten to zero bytes is deactivated; - deleting an indexed file drops its file_sync_state row on the next reindex; - searchFTS and searchVec return at most 256 KiB of a body. --- test/store.test.ts | 163 ++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 162 insertions(+), 1 deletion(-) diff --git a/test/store.test.ts b/test/store.test.ts index 760cdbfc6..e3b7be279 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -9,7 +9,7 @@ import { describe, test, expect, beforeAll, afterAll, beforeEach, afterEach, vi } from "vitest"; import { openDatabase, loadSqliteVec, isBun } from "../src/db.js"; import type { Database } from "../src/db.js"; -import { unlink, mkdtemp, rmdir, writeFile, rm, mkdir, rename, chmod, readFile, symlink } from "node:fs/promises"; +import { unlink, mkdtemp, rmdir, writeFile, rm, mkdir, rename, chmod, readFile, symlink, utimes } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { spawnSync } from "node:child_process"; @@ -52,6 +52,8 @@ import { isDocid, syncConfigToDb, reindexCollection, + removeCollection, + renameCollection, resolveVirtualPath, STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP, @@ -3020,6 +3022,165 @@ describe("Reindex Collection", () => { }); }); +describe("Reindex Collection file sync state (#962)", () => { + const BODY_CAP = 262_144; + + async function collectionDir(prefix: string): Promise { + const dir = join(testDir, `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2)}`); + await mkdir(dir, { recursive: true }); + return dir; + } + + function activeBody(store: Store, collection: string, path: string): string | undefined { + const row = store.db.prepare(` + SELECT content.doc AS body FROM documents d JOIN content ON content.hash = d.hash + WHERE d.collection = ? AND d.path = ? AND d.active = 1 + `).get(collection, path) as { body: string } | undefined; + return row?.body; + } + + function syncRowCount(store: Store, collection: string, path: string): number { + const row = store.db.prepare(` + SELECT COUNT(*) AS n FROM file_sync_state WHERE collection = ? AND relative_path = ? + `).get(collection, path) as { n: number }; + return row.n; + } + + test("reindex trusts an unchanged mtime and size without reading the file", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-fast-path"); + const file = join(dir, "doc.md"); + try { + // A whole-second mtime survives utimes exactly under both runtimes; a + // millisecond Date can come back a fraction lower through Node's seconds. + const mtime = new Date("2026-01-02T03:04:05Z"); + await writeFile(file, "# A\n\nalpha\n"); + await utimes(file, mtime, mtime); + await reindexCollection(store, dir, "**/*.md", "notes"); + // Different bytes of the same size, with the mtime put back. + await writeFile(file, "# A\n\nbravo\n"); + await utimes(file, mtime, mtime); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result).toMatchObject({ indexed: 0, updated: 0, unchanged: 1 }); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("reindex does not read an unchanged file", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-no-read"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + // Stat still works on an unreadable file; a read would fail with EACCES. + await chmod(file, 0o000); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result.skippedFiles).toEqual([]); + expect(result).toMatchObject({ indexed: 0, updated: 0, unchanged: 1, removed: 0 }); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await chmod(file, 0o644); + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("files over 10 MB are skipped with FILE_TOO_LARGE and not indexed", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-too-large"); + try { + await writeFile(join(dir, "big.md"), "a".repeat(10 * 1024 * 1024 + 1)); + await writeFile(join(dir, "small.md"), "# Small\n\nfits\n"); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result.indexed).toBe(1); + expect(result.skippedFiles).toEqual([{ file: "big.md", code: "FILE_TOO_LARGE" }]); + const rows = store.db.prepare(`SELECT COUNT(*) AS n FROM documents WHERE collection = ? AND path = ?`) + .get("notes", "big.md") as { n: number }; + expect(rows.n).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("a previously indexed file that becomes empty is deactivated", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-emptied"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + await writeFile(file, ""); + + await reindexCollection(store, dir, "**/*.md", "notes"); + expect(activeBody(store, "notes", "doc.md")).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("deleting an indexed file removes its sync row on the next reindex", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-deleted"); + try { + await writeFile(join(dir, "doc.md"), "# A\n\nalpha\n"); + await writeFile(join(dir, "keep.md"), "# B\n\nbravo\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + expect(syncRowCount(store, "notes", "doc.md")).toBe(1); + await rm(join(dir, "doc.md")); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result.removed).toBe(1); + expect(syncRowCount(store, "notes", "doc.md")).toBe(0); + expect(syncRowCount(store, "notes", "keep.md")).toBe(1); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("searchFTS returns at most 256 KiB of a document body", async () => { + const store = await createTestStore(); + try { + const body = "# Long\n\nzebracap " + "x".repeat(300 * 1024); + await insertTestDocument(store.db, "docs", { name: "long", body, displayPath: "long.md" }); + + const results = store.searchFTS("zebracap", 5); + expect(results).toHaveLength(1); + expect(results[0]!.body!.length).toBe(BODY_CAP); + expect(results[0]!.body).toBe(body.slice(0, BODY_CAP)); + } finally { + await cleanupTestDb(store); + } + }); + + test("searchVec returns at most 256 KiB of a document body", async () => { + const store = await createTestStore(); + try { + const body = "# Long\n\nvector cap " + "y".repeat(300 * 1024); + const hash = await hashContent(body); + await insertTestDocument(store.db, "docs", { name: "long", body, hash, displayPath: "long.md" }); + store.ensureVecTable(3); + store.insertEmbedding(hash, 0, 0, new Float32Array([1, 0, 0]), "cap-model", new Date().toISOString()); + + const results = await store.searchVec("q", "cap-model", 5, undefined, undefined, [1, 0, 0]); + expect(results).toHaveLength(1); + expect(results[0]!.body!.length).toBe(BODY_CAP); + expect(results[0]!.body).toBe(body.slice(0, BODY_CAP)); + } finally { + await cleanupTestDb(store); + } + }); +}); + // ============================================================================= // Index Status Tests // ============================================================================= From d4567fa3ecb48f79c37d471eb07d67a8618efc0d Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:52:30 -0500 Subject: [PATCH 46/82] fix(store): trust a file_sync_state row only while its document stands #962's fast path trusts a cached mtime and size, and its hash-match branch trusts a cached hash, without checking the document the row points at. Removing a collection or renaming it away leaves its rows behind, so adding the collection back (or adding the old name at another path) counted every file as unchanged and left it with no active document. The sync-state lookup now keeps a row only when its document is still the active row for that collection, path and content hash. Any other row counts as absent: the file is read, indexed and its row rewritten. The collection remove and rename paths are unchanged. --- src/store.ts | 17 ++++++++++--- test/store.test.ts | 63 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 77 insertions(+), 3 deletions(-) diff --git a/src/store.ts b/src/store.ts index 5e4b716bf..6fe5e7c34 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1642,9 +1642,20 @@ type FileSyncStateRow = { function getFileSyncStateMap(db: Database, collectionName: string): Map { try { - const stmt = db.prepare( - `SELECT relative_path, mtime_ms, size, content_hash, document_id FROM file_sync_state WHERE collection = ?` - ); + // A row is trusted only while its document is still the active row for + // this collection, path and content. Removing a collection, or renaming it + // away, leaves rows behind; trusting those would count a file that has no + // active document as unchanged. A distrusted row is rewritten on reindex. + const stmt = db.prepare(` + SELECT s.relative_path, s.mtime_ms, s.size, s.content_hash, s.document_id + FROM file_sync_state s + JOIN documents d ON d.id = s.document_id + WHERE s.collection = ? + AND d.active = 1 + AND d.collection = s.collection + AND d.path = s.relative_path + AND d.hash = s.content_hash + `); const map = new Map(); // Large-result query: use iterate() to stream rows instead of .all() materializing at once for (const r of stmt.iterate(collectionName) as IterableIterator) { diff --git a/test/store.test.ts b/test/store.test.ts index e3b7be279..9508f39fc 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3147,6 +3147,69 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("removing a collection and adding it back re-indexes its files", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-readd"); + try { + await writeFile(join(dir, "doc.md"), "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + removeCollection(store.db, "notes"); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result.indexed).toBe(1); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("removing and re-adding a collection re-indexes a file whose mtime moved but content did not", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-readd-touched"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await utimes(file, new Date("2026-01-02T03:04:05Z"), new Date("2026-01-02T03:04:05Z")); + await reindexCollection(store, dir, "**/*.md", "notes"); + removeCollection(store.db, "notes"); + await utimes(file, new Date("2026-01-02T03:05:05Z"), new Date("2026-01-02T03:05:05Z")); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result.indexed).toBe(1); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("renaming a collection and adding the old name at another path indexes both", async () => { + const store = await createTestStore(); + const first = await collectionDir("sync-rename-first"); + const second = await collectionDir("sync-rename-second"); + const mtime = new Date("2026-01-02T03:04:05Z"); + try { + // Same relative path, size and mtime in both directories. + await writeFile(join(first, "doc.md"), "# A\n\nalpha\n"); + await writeFile(join(second, "doc.md"), "# A\n\nbravo\n"); + await utimes(join(first, "doc.md"), mtime, mtime); + await utimes(join(second, "doc.md"), mtime, mtime); + await reindexCollection(store, first, "**/*.md", "notes"); + renameCollection(store.db, "notes", "archive"); + + await reindexCollection(store, first, "**/*.md", "archive"); + const result = await reindexCollection(store, second, "**/*.md", "notes"); + expect(result.indexed).toBe(1); + expect(activeBody(store, "archive", "doc.md")).toContain("alpha"); + expect(activeBody(store, "notes", "doc.md")).toContain("bravo"); + } finally { + await rm(first, { recursive: true, force: true }); + await rm(second, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("searchFTS returns at most 256 KiB of a document body", async () => { const store = await createTestStore(); try { From 1f75eb7e4a261d13d9f4fa225d26c82e156c5ac9 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:07:31 -0500 Subject: [PATCH 47/82] test(store): pin the active and content clauses of the sync-row check The sync-row trust check requires the row's document to be active, at the same collection and path, with the same content hash. The rename and re-add tests only exercise the collection and path clauses. These cover the other two: a document deactivated behind the index's back, and one pointed at other content, each with its file and sync row left unchanged, must be re-read and restored on the next reindex. --- test/store.test.ts | 38 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/test/store.test.ts b/test/store.test.ts index 9508f39fc..f57f9070e 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3210,6 +3210,44 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("a file whose document was deactivated elsewhere is re-indexed", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-deactivated"); + try { + await writeFile(join(dir, "doc.md"), "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + // Unchanged file and sync row; only the document went inactive. + store.deactivateDocument("notes", "doc.md"); + + await reindexCollection(store, dir, "**/*.md", "notes"); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("a file whose document now points at other content is re-read", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-rehashed"); + try { + await writeFile(join(dir, "doc.md"), "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + // Unchanged file and sync row; only the document's content changed. + const now = new Date().toISOString(); + store.insertContent("elsewhere-hash", "# A\n\nwritten elsewhere\n", now); + const doc = store.db.prepare(`SELECT id FROM documents WHERE collection = ? AND path = ?`) + .get("notes", "doc.md") as { id: number }; + store.updateDocument(doc.id, "A", "elsewhere-hash", now); + + await reindexCollection(store, dir, "**/*.md", "notes"); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("searchFTS returns at most 256 KiB of a document body", async () => { const store = await createTestStore(); try { From 03691f206ad9651b614cd38cb704acc1d3b20b70 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:54:09 -0500 Subject: [PATCH 48/82] fix(store): re-extract stale metadata for files the fast path skips #962's fast path skips the read, and with it the metadata sync, for every file whose mtime and size match. An index built before the metadata schema, or one whose extraction version is bumped, then never gets current metadata for unchanged files, which drops them from metadata-filtered search. The upstream test that backfills documents indexed before the metadata schema failed on #962 for this reason. The fast path now also requires a current extraction row, using the existing isDocumentMetadataCurrent check (now exported). A file without one is read and re-extracted through the hash-match branch with onlyIfStale, as upstream does for unchanged content. --- src/metadata-store.ts | 3 ++- src/store.ts | 7 ++++++- test/metadata-store.test.ts | 15 +++++++++++++++ 3 files changed, 23 insertions(+), 2 deletions(-) diff --git a/src/metadata-store.ts b/src/metadata-store.ts index d97e273b5..456740eb7 100644 --- a/src/metadata-store.ts +++ b/src/metadata-store.ts @@ -162,7 +162,8 @@ export function replaceDocumentMetadata(db: Database, documentId: number, extrac replace(); } -function isDocumentMetadataCurrent(db: Database, documentId: number): boolean { +/** Whether a document has a metadata extraction row at the current extraction version. */ +export function isDocumentMetadataCurrent(db: Database, documentId: number): boolean { const row = db.prepare(`SELECT extraction_version FROM document_metadata WHERE document_id = ?`) .get(documentId) as { extraction_version: number } | undefined; return row?.extraction_version === METADATA_EXTRACTION_VERSION; diff --git a/src/store.ts b/src/store.ts index 6fe5e7c34..b398dccae 100644 --- a/src/store.ts +++ b/src/store.ts @@ -42,6 +42,7 @@ import { compileMetadataFilter, type MetadataFilter } from "./metadata-filter.js import { initializeMetadataSchema, syncDocumentMetadata, + isDocumentMetadataCurrent, countDocumentsPendingMetadata, getMetadataByFilepath, listMetadataCollectionSummaries, @@ -1786,7 +1787,11 @@ export async function reindexCollection( // Fast-path: stat matches cached sync state — skip read entirely const cached = syncStateMap.get(path); - if (cached && cached.mtime_ms === Math.floor(mtimeMs) && cached.size === size) { + // Missing or stale metadata (an index from before the metadata schema, or + // an extraction-version bump) needs the content, so such a file is read and + // re-extracted through the hash-match branch below. + if (cached && cached.mtime_ms === Math.floor(mtimeMs) && cached.size === size + && isDocumentMetadataCurrent(db, cached.document_id)) { unchanged++; processed++; options?.onProgress?.({ file: relativeFile, current: processed, total }); diff --git a/test/metadata-store.test.ts b/test/metadata-store.test.ts index a7e11becb..9a0c8301b 100644 --- a/test/metadata-store.test.ts +++ b/test/metadata-store.test.ts @@ -228,6 +228,21 @@ describe("reindexCollection metadata synchronization", () => { expect(getMetadataForPath("doc.md")).toEqual({ status: "ok" }); }); + test("re-extracts an unchanged file whose metadata came from an older extraction version", async () => { + await writeFile(join(collectionDir, "doc.md"), buildDoc("qmd:\n metadata:\n status: ok\n", "# Doc\n")); + await reindex(); + + store.db.prepare(`UPDATE document_metadata SET extraction_version = 0`).run(); + expect(countDocumentsPendingMetadata(store.db)).toBe(1); + + const result = await reindex(); + expect(result.unchanged).toBe(1); + const row = store.db.prepare(`SELECT extraction_version FROM document_metadata`).get() as { extraction_version: number }; + expect(row.extraction_version).toBe(METADATA_EXTRACTION_VERSION); + expect(countDocumentsPendingMetadata(store.db)).toBe(0); + expect(getMetadataForPath("doc.md")).toEqual({ status: "ok" }); + }); + test("counts extraction errors without aborting the collection", async () => { await writeFile(join(collectionDir, "bad.md"), buildDoc("qmd:\n metadata:\n mixed: [1, two]\n", "# Bad\n")); await writeFile(join(collectionDir, "good.md"), buildDoc("qmd:\n metadata:\n status: ok\n", "# Good\n")); From 92ac75a0108ddfe419c5b8269eeda5ec1dd75ec9 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 09:55:07 -0500 Subject: [PATCH 49/82] fix(store): deactivate a document whose file grows past 10 MB #962 skips a file over 10 MB before reading it, but the file is already marked seen by then, so the orphan pass never deactivates it. A file that was indexed and then grew past the limit kept its old content active and searchable, along with its sync row. The size guard now retires a previously indexed file the same way #962 already retires one that became empty: deactivate the document and drop its sync row. Both branches share one helper. --- src/store.ts | 26 ++++++++++++++++++++------ test/store.test.ts | 19 +++++++++++++++++++ 2 files changed, 39 insertions(+), 6 deletions(-) diff --git a/src/store.ts b/src/store.ts index b398dccae..d37e72d84 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1697,6 +1697,24 @@ function deleteFileSyncStateForCollection(db: Database, collectionName: string, */ const REINDEX_MAX_FILE_SIZE = 10 * 1024 * 1024; // 10MB +/** + * Deactivate a previously indexed file that can no longer be indexed (it + * became empty or grew past REINDEX_MAX_FILE_SIZE) and drop its sync row, so + * its old content stops being searchable. + */ +function retireIndexedFile( + db: Database, + collectionName: string, + path: string, + livePaths: ReadonlySet, + syncStateMap: Map, +): void { + if (!findOrMigrateLegacyDocument(db, collectionName, path, livePaths)) return; + deactivateDocument(db, collectionName, path); + deleteFileSyncStateForCollection(db, collectionName, path); + syncStateMap.delete(path); +} + /** * Re-index a single collection by scanning the filesystem and updating the database. * Uses mtime+size fast-path (file_sync_state) to avoid re-reading unchanged files. @@ -1779,6 +1797,7 @@ export async function reindexCollection( // Skip large files (>10MB) — prevents OOM if (size > REINDEX_MAX_FILE_SIZE) { + retireIndexedFile(db, collectionName, path, livePaths, syncStateMap); processed++; skippedFiles.push({ file: relativeFile, code: "FILE_TOO_LARGE" }); options?.onProgress?.({ file: relativeFile, current: processed, total }); @@ -1814,12 +1833,7 @@ export async function reindexCollection( if (!content.trim()) { // Empty file — if previously indexed, deactivate it (treat as removed) - const existingEmpty = findOrMigrateLegacyDocument(db, collectionName, path, livePaths); - if (existingEmpty) { - deactivateDocument(db, collectionName, path); - deleteFileSyncStateForCollection(db, collectionName, path); - syncStateMap.delete(path); - } + retireIndexedFile(db, collectionName, path, livePaths, syncStateMap); processed++; options?.onProgress?.({ file: relativeFile, current: processed, total }); continue; diff --git a/test/store.test.ts b/test/store.test.ts index f57f9070e..66b1b059d 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3110,6 +3110,25 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("a previously indexed file that grows past 10 MB is reported and deactivated", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-grown"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + await writeFile(file, "# A\n\n" + "a".repeat(10 * 1024 * 1024)); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result.skippedFiles).toEqual([{ file: "doc.md", code: "FILE_TOO_LARGE" }]); + expect(activeBody(store, "notes", "doc.md")).toBeUndefined(); + expect(syncRowCount(store, "notes", "doc.md")).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("a previously indexed file that becomes empty is deactivated", async () => { const store = await createTestStore(); const dir = await collectionDir("sync-emptied"); From c6bd2c0d8ca455c4f82610d00b1b94364b010c25 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 10:00:48 -0500 Subject: [PATCH 50/82] test(store): cover #1000's expansion dedupe on fresh output and vsearch #1000's own test covers a cached row through hybridQuery, using a legacy vectors_vec stub. These cover the other two paths through ensureVecTable, so they hold once the vector layout changes: - a model that emits hyde "g" three times and vec "v" twice yields two expansions from expandQuery, and only those two are cached; - a cached row of 12 identical hyde lines and 2 identical vec lines makes vectorSearchQuery search three times (the query plus two distinct expansions) instead of fifteen. --- test/store.test.ts | 45 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 45 insertions(+) diff --git a/test/store.test.ts b/test/store.test.ts index 2e2f7b4e6..b407a9ed7 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -1400,6 +1400,51 @@ describe("Query expansion cache (#818)", () => { await cleanupTestDb(store); } }); + + test("expandQuery drops repeated lines from fresh model output before caching them", async () => { + const store = await createTestStore(); + const generateModelName = "dedupe-generate-model"; + store.llm = { + generateModelName, + expandQuery: async () => [ + { type: "hyde", text: "g" }, { type: "hyde", text: "g" }, { type: "hyde", text: "g" }, + { type: "vec", text: "v" }, { type: "vec", text: "v" }, + ], + } as any; + try { + const expanded = await store.expandQuery("q"); + expect(expanded).toEqual([{ type: "hyde", query: "g" }, { type: "vec", query: "v" }]); + const cached = store.getCachedResult(getCacheKey("expandQuery", { query: "q", model: generateModelName })); + expect(JSON.parse(cached!)).toHaveLength(2); + } finally { + await cleanupTestDb(store); + } + }); + + test("vectorSearchQuery searches each distinct cached expansion once", async () => { + const store = await createTestStore(); + const embedModel = "dedupe-embed-model"; + store.ensureVecTable(3); + store.llm = { embedModelName: embedModel } as any; + const searchVecSpy = vi.fn(async () => [] as SearchResult[]); + store.searchVec = searchVecSpy as any; + try { + const cached = [ + ...Array.from({ length: 12 }, () => ({ type: "hyde", query: "g" })), + { type: "vec", query: "v" }, + { type: "vec", query: "v" }, + ]; + store.setCachedResult(getCacheKey("expandQuery", { query: "q", model: DEFAULT_QUERY_MODEL }), JSON.stringify(cached)); + + await vectorSearchQuery(store, "q", { limit: 5 }); + + // The original query plus one search per distinct expansion. + expect(searchVecSpy).toHaveBeenCalledTimes(3); + expect(searchVecSpy.mock.calls.map(call => call[0])).toEqual(["q", "g", "v"]); + } finally { + await cleanupTestDb(store); + } + }); }); From bcf8b18b260dd93c0990c0902146f911557582e2 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 10:58:09 -0500 Subject: [PATCH 51/82] test(search): cover #953's collection scope combined with a metadata filter #953's tests cover each scope on its own. This one combines them: a small collection holds a draft that outranks the published target, a large collection's noise outranks both, and a search scoped to the small collection with a status filter returns only the target. A guard checks the scoped path still caps bodies at 256 KiB. --- test/metadata-search.test.ts | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index 8b8c3f864..30168d86d 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -319,3 +319,33 @@ describe("structuredSearch with metadata filter", () => { expect(byPath.get("notes/b.md")).toEqual({}); }); }); + +describe("searchFTS with a collection scope and a metadata filter together", () => { + const published: MetadataFilter = { key: "status", operator: "eq", value: "published" }; + + test("returns the in-scope document the filter admits, past both the window and a stronger draft", async () => { + // Every noise document outranks both small-collection documents globally, + // and the draft outranks the target inside the small collection. + for (let i = 0; i < 50; i++) { + await insertDoc("large", `noise-${i}.md`, `# N${i}\n\nalpha alpha alpha`, { status: "published" }); + } + await insertDoc("small", "draft.md", "# Draft\n\nalpha alpha", { status: "draft" }); + await insertDoc( + "small", + "target.md", + `# Target\n\n${"Unrelated prose. ".repeat(40)}One weaker mention of alpha.`, + { status: "published" }, + ); + + const results = searchFTS(store.db, "alpha", 1, "small", published); + expect(results.map(r => r.displayPath)).toEqual(["small/target.md"]); + }); + + test("keeps the 256 KiB body cap on the scoped path", async () => { + await insertDoc("small", "long.md", `# Long\n\nalpha ${"z".repeat(300 * 1024)}`, { status: "published" }); + + const results = searchFTS(store.db, "alpha", 5, "small", published); + expect(results).toHaveLength(1); + expect(results[0]!.body!.length).toBe(262_144); + }); +}); From 87e7b3f81d5907e3fb4e2adb41e82d804d2722de Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:06:51 -0500 Subject: [PATCH 52/82] test(store): check that the SQLite heap limit binds in the budget workers The doctor and legacy-adoption budget tests only prove their 32 MiB and 16 MiB claims where SQLite enforces PRAGMA hard_heap_limit. Node's better-sqlite3 accepts the pragma without enforcing it, and a Bun or SQLite build that stopped tracking memory would let both tests pass on any query. Each worker now allocates past its limit after setting it and reports whether SQLite refused (limitBinds). On Bun under Linux, where Bun's bundled SQLite tracks memory, the tests require it. --- test/_helpers/doctor-vector-check-worker.ts | 3 ++- test/_helpers/heap-limit.ts | 16 ++++++++++++++++ .../legacy-adoption-shared-body-worker.ts | 3 ++- test/doctor-vector-sampling.test.ts | 5 ++++- test/store.test.ts | 5 ++++- 5 files changed, 28 insertions(+), 4 deletions(-) create mode 100644 test/_helpers/heap-limit.ts diff --git a/test/_helpers/doctor-vector-check-worker.ts b/test/_helpers/doctor-vector-check-worker.ts index 18a7e6be1..d0d23af05 100644 --- a/test/_helpers/doctor-vector-check-worker.ts +++ b/test/_helpers/doctor-vector-check-worker.ts @@ -1,6 +1,7 @@ import { createStore } from "../../src/store.js"; import { checkEmbeddingVectorSamples } from "../../src/cli/qmd.js"; import { LlamaCpp, setDefaultLlamaCpp } from "../../src/llm.js"; +import { heapLimitBinds } from "./heap-limit.js"; // Every stored vector is [1, 0] and every re-embedded chunk returns [1, 0], so // the check passes whenever it can sample, read and chunk the bodies. @@ -29,7 +30,7 @@ try { store.db.exec("PRAGMA temp_store = MEMORY"); store.db.exec("PRAGMA hard_heap_limit = 33554432"); const result = await checkEmbeddingVectorSamples(store.db, "model", "current"); - console.log(JSON.stringify(result)); + console.log(JSON.stringify({ ...result, limitBinds: heapLimitBinds(store.db, 33554432) })); } finally { setDefaultLlamaCpp(null); store.close(); diff --git a/test/_helpers/heap-limit.ts b/test/_helpers/heap-limit.ts new file mode 100644 index 000000000..0a01a7ccc --- /dev/null +++ b/test/_helpers/heap-limit.ts @@ -0,0 +1,16 @@ +import type { Database } from "../../src/db.js"; + +/** + * Whether SQLite enforces the hard heap limit just set on `db`: an allocation + * past it fails with SQLITE_NOMEM only when SQLite tracks memory. Bun's bundled + * SQLite on Linux does; better-sqlite3 (DEFAULT_MEMSTATUS=0) does not, so a + * heap-budget test proves its budget only where this returns true. + */ +export function heapLimitBinds(db: Database, limitBytes: number): boolean { + try { + db.prepare(`SELECT length(randomblob(?)) AS n`).get(limitBytes + 8 * 1024 * 1024); + return false; + } catch { + return true; + } +} diff --git a/test/_helpers/legacy-adoption-shared-body-worker.ts b/test/_helpers/legacy-adoption-shared-body-worker.ts index 3962a769d..02a0a0664 100644 --- a/test/_helpers/legacy-adoption-shared-body-worker.ts +++ b/test/_helpers/legacy-adoption-shared-body-worker.ts @@ -12,6 +12,7 @@ import { mkdtempSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { createStore, maybeAdoptLegacyEmbeddingFingerprint } from "../../src/store.ts"; +import { heapLimitBinds } from "./heap-limit.ts"; const model = "model"; const dir = mkdtempSync(join(tmpdir(), "qmd-legacy-shared-body-")); @@ -36,7 +37,7 @@ try { store.db.exec("PRAGMA temp_store = MEMORY"); store.db.exec("PRAGMA hard_heap_limit = 16777216"); const result = await maybeAdoptLegacyEmbeddingFingerprint(store, model); - console.log(JSON.stringify({ checked: result.checked, adopted: result.adopted })); + console.log(JSON.stringify({ checked: result.checked, adopted: result.adopted, limitBinds: heapLimitBinds(store.db, 16777216) })); } finally { store.close(); rmSync(dir, { recursive: true, force: true }); diff --git a/test/doctor-vector-sampling.test.ts b/test/doctor-vector-sampling.test.ts index faff96c8a..f912f350d 100644 --- a/test/doctor-vector-sampling.test.ts +++ b/test/doctor-vector-sampling.test.ts @@ -118,6 +118,9 @@ describe("doctor vector sampling", () => { expect(result.error).toBeUndefined(); expect(result.stderr).toBe(""); expect(result.status).toBe(0); - expect(JSON.parse(result.stdout)).toEqual({ ok: true, details: "3 sampled chunks reproduce stored vectors" }); + const out = JSON.parse(result.stdout); + expect(out).toMatchObject({ ok: true, details: "3 sampled chunks reproduce stored vectors" }); + // The budget only binds where SQLite tracks memory; on Bun under Linux it must. + if (isBun && process.platform === "linux") expect(out.limitBinds).toBe(true); }); }); diff --git a/test/store.test.ts b/test/store.test.ts index a559c5724..760cdbfc6 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -4350,7 +4350,10 @@ describe("Embedding batching", () => { expect(result.error).toBeUndefined(); expect(result.stderr).toBe(""); expect(result.status).toBe(0); - expect(JSON.parse(result.stdout)).toEqual({ checked: true, adopted: 1 }); + const out = JSON.parse(result.stdout); + expect(out).toMatchObject({ checked: true, adopted: 1 }); + // The budget only binds where SQLite tracks memory; on Bun under Linux it must. + if (isBun && process.platform === "linux") expect(out.limitBinds).toBe(true); }); test("generateEmbeddings flushes batches when maxDocsPerBatch is reached", async () => { From aaf1911ae2638fe26036c6ec64618c67e4bc6182 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:14:00 -0500 Subject: [PATCH 53/82] perf(store): read metadata currency in the sync-state query The fast path checked each unchanged file's metadata extraction with its own query, one per file on every update. The sync-state query now joins document_metadata once and returns whether each row's document has an extraction at the current version, and isDocumentMetadataCurrent goes back to being private to metadata-store. --- src/metadata-store.ts | 3 +-- src/store.ts | 12 +++++++----- 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/src/metadata-store.ts b/src/metadata-store.ts index 456740eb7..d97e273b5 100644 --- a/src/metadata-store.ts +++ b/src/metadata-store.ts @@ -162,8 +162,7 @@ export function replaceDocumentMetadata(db: Database, documentId: number, extrac replace(); } -/** Whether a document has a metadata extraction row at the current extraction version. */ -export function isDocumentMetadataCurrent(db: Database, documentId: number): boolean { +function isDocumentMetadataCurrent(db: Database, documentId: number): boolean { const row = db.prepare(`SELECT extraction_version FROM document_metadata WHERE document_id = ?`) .get(documentId) as { extraction_version: number } | undefined; return row?.extraction_version === METADATA_EXTRACTION_VERSION; diff --git a/src/store.ts b/src/store.ts index d37e72d84..e24d90e93 100644 --- a/src/store.ts +++ b/src/store.ts @@ -42,7 +42,6 @@ import { compileMetadataFilter, type MetadataFilter } from "./metadata-filter.js import { initializeMetadataSchema, syncDocumentMetadata, - isDocumentMetadataCurrent, countDocumentsPendingMetadata, getMetadataByFilepath, listMetadataCollectionSummaries, @@ -1639,6 +1638,8 @@ type FileSyncStateRow = { size: number; content_hash: string; document_id: number; + /** 1 when the document has a metadata extraction at the current version. */ + metadata_current: number; }; function getFileSyncStateMap(db: Database, collectionName: string): Map { @@ -1648,9 +1649,11 @@ function getFileSyncStateMap(db: Database, collectionName: string): Map(); // Large-result query: use iterate() to stream rows instead of .all() materializing at once - for (const r of stmt.iterate(collectionName) as IterableIterator) { + for (const r of stmt.iterate(METADATA_EXTRACTION_VERSION, collectionName) as IterableIterator) { map.set(r.relative_path, r); } return map; @@ -1809,8 +1812,7 @@ export async function reindexCollection( // Missing or stale metadata (an index from before the metadata schema, or // an extraction-version bump) needs the content, so such a file is read and // re-extracted through the hash-match branch below. - if (cached && cached.mtime_ms === Math.floor(mtimeMs) && cached.size === size - && isDocumentMetadataCurrent(db, cached.document_id)) { + if (cached && cached.mtime_ms === Math.floor(mtimeMs) && cached.size === size && cached.metadata_current) { unchanged++; processed++; options?.onProgress?.({ file: relativeFile, current: processed, total }); From 5660ad6255ccf96b01aff55f6cab10e2e5e8610e Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:14:29 -0500 Subject: [PATCH 54/82] test(store): cover a touched file whose content is unchanged When only the mtime moves and the content hash still matches, #962's reindex refreshes the sync row's mtime and counts the file unchanged instead of re-indexing it. The test checks the counts and the stored mtime, then makes the file unreadable to show the next reindex takes the fast path on the refreshed row. --- test/store.test.ts | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/test/store.test.ts b/test/store.test.ts index 66b1b059d..0e8a13b26 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3091,6 +3091,36 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("a touched file whose content is unchanged refreshes its sync row without re-indexing", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-touch"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await utimes(file, new Date("2026-01-02T03:04:05Z"), new Date("2026-01-02T03:04:05Z")); + await reindexCollection(store, dir, "**/*.md", "notes"); + const touched = new Date("2026-01-02T03:05:05Z"); + await utimes(file, touched, touched); + + const first = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(first).toMatchObject({ indexed: 0, updated: 0, unchanged: 1, removed: 0 }); + const row = store.db.prepare(`SELECT mtime_ms FROM file_sync_state WHERE collection = ? AND relative_path = ?`) + .get("notes", "doc.md") as { mtime_ms: number }; + expect(row.mtime_ms).toBe(touched.getTime()); + + // The refreshed row puts the file back on the fast path: a read would now report EACCES. + await chmod(file, 0o000); + const second = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(second.skippedFiles).toEqual([]); + expect(second).toMatchObject({ indexed: 0, updated: 0, unchanged: 1, removed: 0 }); + expect(activeBody(store, "notes", "doc.md")).toContain("alpha"); + } finally { + await chmod(file, 0o644); + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("files over 10 MB are skipped with FILE_TOO_LARGE and not indexed", async () => { const store = await createTestStore(); const dir = await collectionDir("sync-too-large"); From 133bc4f43d6d3ad88c3e5892bb4270439281429a Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:16:24 -0500 Subject: [PATCH 55/82] fix(cli): skip files over 10 MB when adding a collection #962 skips files over 10 MB in reindexCollection, which qmd update uses, but qmd collection add indexes through its own loop and still read and indexed them. That loop now checks the file size first and reports a larger file as FILE_TOO_LARGE, using the same limit, which store.ts now exports. --- src/cli/qmd.ts | 9 +++++++++ src/store.ts | 2 +- test/cli.test.ts | 16 ++++++++++++++++ 3 files changed, 26 insertions(+), 1 deletion(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 2c074c609..6c5d62d34 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -81,6 +81,7 @@ import { createStore, getDefaultDbPath, reindexCollection, + REINDEX_MAX_FILE_SIZE, generateEmbeddings, maybeAdoptLegacyEmbeddingFingerprint, syncConfigToDb, @@ -2141,6 +2142,14 @@ async function indexFiles(pwd?: string, globPattern: string = DEFAULT_GLOB, coll progress.set((processed / total) * 100); continue; } + let tooLarge = false; + try { tooLarge = statSync(filepath).size > REINDEX_MAX_FILE_SIZE; } catch { /* the read below reports it */ } + if (tooLarge) { + processed++; + skippedFiles.push({ file: relativeFile, code: "FILE_TOO_LARGE" }); + progress.set((processed / total) * 100); + continue; + } seenPaths.add(path); let content: string; diff --git a/src/store.ts b/src/store.ts index e24d90e93..73ac7cba3 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1698,7 +1698,7 @@ function deleteFileSyncStateForCollection(db: Database, collectionName: string, /** * Maximum file size to index — prevents OOM on accidental binary inclusion. */ -const REINDEX_MAX_FILE_SIZE = 10 * 1024 * 1024; // 10MB +export const REINDEX_MAX_FILE_SIZE = 10 * 1024 * 1024; // 10MB /** * Deactivate a previously indexed file that can no longer be indexed (it diff --git a/test/cli.test.ts b/test/cli.test.ts index 8fdbcf50c..2b156104d 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -687,6 +687,22 @@ describe("CLI Add Command", () => { } }); + test("collection add skips files over 10 MB", async () => { + const env = await createIsolatedTestEnv("skip-too-large"); + const collectionDir = join(testDir, `skip-too-large-${testCounter}`); + await mkdir(collectionDir, { recursive: true }); + await writeFile(join(collectionDir, "good.md"), "alpha\n"); + await writeFile(join(collectionDir, "big.md"), "a".repeat(10 * 1024 * 1024 + 1)); + + const { stdout, stderr, exitCode } = await runQmd( + ["collection", "add", collectionDir, "--name", "skip-too-large"], + { dbPath: env.dbPath, configDir: env.configDir, cwd: collectionDir }, + ); + expect(exitCode).toBe(0); + expect(stdout).toContain("Indexed: 1 new"); + expect(stderr).toContain("big.md (FILE_TOO_LARGE)"); + }); + test("can recreate collection with remove and add", async () => { // First add await runQmd(["collection", "add", "."]); From e7f36e90debeb878f1ea9343fd15fa7cc55c5025 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:25:14 -0500 Subject: [PATCH 56/82] test(store): cover #1000's dedupe on older cache rows and keyword lines Two paths through #1000's dedupe had no test. A cached row written in the older shape, which stores each line as `text`, now searches each distinct expansion once through vectorSearchQuery. And hybridQuery, given a cached row with the same lex line three times, runs one keyword search for it after the probe of the original query. --- test/store.test.ts | 51 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 51 insertions(+) diff --git a/test/store.test.ts b/test/store.test.ts index b407a9ed7..92c030d28 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -1445,6 +1445,57 @@ describe("Query expansion cache (#818)", () => { await cleanupTestDb(store); } }); + + test("vectorSearchQuery searches each distinct expansion once from a cached row in the older text shape", async () => { + const store = await createTestStore(); + const embedModel = "dedupe-embed-model"; + store.ensureVecTable(3); + store.llm = { embedModelName: embedModel } as any; + const searchVecSpy = vi.fn(async () => [] as SearchResult[]); + store.searchVec = searchVecSpy as any; + try { + // Rows written before the cache stored `query` carry the line as `text`. + const cached = [ + ...Array.from({ length: 12 }, () => ({ type: "hyde", text: "g" })), + { type: "vec", text: "v" }, + { type: "vec", text: "v" }, + ]; + store.setCachedResult(getCacheKey("expandQuery", { query: "q", model: DEFAULT_QUERY_MODEL }), JSON.stringify(cached)); + + await vectorSearchQuery(store, "q", { limit: 5 }); + + expect(searchVecSpy.mock.calls.map(call => call[0])).toEqual(["q", "g", "v"]); + } finally { + await cleanupTestDb(store); + } + }); + + test("hybridQuery runs one keyword search per distinct cached lex line", async () => { + const store = await createTestStore(); + const embedModel = "dedupe-embed-model"; + store.ensureVecTable(3); + store.llm = { + embedModelName: embedModel, + embedBatch: async (texts: string[]) => texts.map(() => ({ embedding: [1, 2, 3], model: embedModel })), + } as any; + store.searchVec = vi.fn(async () => [] as SearchResult[]) as any; + const searchFTSSpy = vi.fn(() => [] as SearchResult[]); + store.searchFTS = searchFTSSpy as any; + try { + const cached = [ + { type: "lex", query: "k" }, { type: "lex", query: "k" }, { type: "lex", query: "k" }, + { type: "vec", query: "v" }, + ]; + store.setCachedResult(getCacheKey("expandQuery", { query: "q", model: DEFAULT_QUERY_MODEL }), JSON.stringify(cached)); + + await hybridQuery(store, "q", { limit: 5, minScore: 0, skipRerank: true, intent: "x" }); + + // The original query's probe, then the distinct lex line once. + expect(searchFTSSpy.mock.calls.map(call => call[0])).toEqual(["q", "k"]); + } finally { + await cleanupTestDb(store); + } + }); }); From 74fc668ee8fc21c2f8e7acba8da9951b20a79d5b Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:50:07 -0500 Subject: [PATCH 57/82] fix(store): drop sync rows with their collection and move them on rename Removing a collection left its file_sync_state rows behind for good, and renaming one left them under the old name. The sync-row check already ignored both, so nothing read them, but they were never purged, and after a rename every file was read and hashed once more under the new name. removeCollection now deletes the collection's rows and renameCollection moves them, first clearing any rows already under the new name, since no collection owns those. --- src/store.ts | 5 +++++ test/store.test.ts | 40 ++++++++++++++++++++++++++++++++++++++++ 2 files changed, 45 insertions(+) diff --git a/src/store.ts b/src/store.ts index 73ac7cba3..c35be02b2 100644 --- a/src/store.ts +++ b/src/store.ts @@ -3949,6 +3949,7 @@ export function listCollections(db: Database): { name: string; pwd: string; glob export function removeCollection(db: Database, collectionName: string): { deletedDocs: number; cleanedHashes: number } { // Delete documents from database const docResult = db.prepare(`DELETE FROM documents WHERE collection = ?`).run(collectionName); + db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(collectionName); // Clean up orphaned content hashes const cleanupResult = db.prepare(` @@ -3973,6 +3974,10 @@ export function renameCollection(db: Database, oldName: string, newName: string) // Update all documents with the new collection name in database db.prepare(`UPDATE documents SET collection = ? WHERE collection = ?`) .run(newName, oldName); + // The documents keep their ids and paths, so their sync rows stay valid + // under the new name. Rows already under it belong to no collection. + db.prepare(`DELETE FROM file_sync_state WHERE collection = ?`).run(newName); + db.prepare(`UPDATE file_sync_state SET collection = ? WHERE collection = ?`).run(newName, oldName); // Rename in store_collections renameStoreCollection(db, oldName, newName); diff --git a/test/store.test.ts b/test/store.test.ts index 0e8a13b26..eb6772c1d 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3233,6 +3233,46 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("removing a collection deletes its sync rows", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-remove-rows"); + try { + await writeFile(join(dir, "doc.md"), "# A\n\nalpha\n"); + await reindexCollection(store, dir, "**/*.md", "notes"); + expect(syncRowCount(store, "notes", "doc.md")).toBe(1); + + removeCollection(store.db, "notes"); + expect(syncRowCount(store, "notes", "doc.md")).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("renaming a collection moves its sync rows, so the new name keeps the fast path", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-rename-rows"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await utimes(file, new Date("2026-01-02T03:04:05Z"), new Date("2026-01-02T03:04:05Z")); + await reindexCollection(store, dir, "**/*.md", "notes"); + renameCollection(store.db, "notes", "archive"); + expect(syncRowCount(store, "notes", "doc.md")).toBe(0); + expect(syncRowCount(store, "archive", "doc.md")).toBe(1); + + // A read would now report EACCES. + await chmod(file, 0o000); + const result = await reindexCollection(store, dir, "**/*.md", "archive"); + expect(result.skippedFiles).toEqual([]); + expect(result).toMatchObject({ indexed: 0, updated: 0, unchanged: 1, removed: 0 }); + } finally { + await chmod(file, 0o644); + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("renaming a collection and adding the old name at another path indexes both", async () => { const store = await createTestStore(); const first = await collectionDir("sync-rename-first"); From 94840c870b28b518a8c13e40f8db7c90273458a9 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:55:14 -0500 Subject: [PATCH 58/82] fix(store): do not trust a sync row for a file modified during the pass #962's fast path trusts a file whose mtime and size match its sync row. On a filesystem with a 1-2 s mtime granule, a same-size rewrite in the same granule as the pass that read the file keeps both, so the new content was never indexed until the file changed again. A file whose mtime is within two seconds of the stat that read it now gets a sync row the fast path never matches. The next update reads it once more, through the hash-match branch when nothing changed, and stores a trusted row once the mtime is older, as git does for racily clean index entries. The no-read test now backdates its file first. --- src/store.ts | 17 +++++++++++++++-- test/store.test.ts | 25 +++++++++++++++++++++++++ 2 files changed, 40 insertions(+), 2 deletions(-) diff --git a/src/store.ts b/src/store.ts index c35be02b2..ae1e4cb96 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1695,6 +1695,17 @@ function deleteFileSyncStateForCollection(db: Database, collectionName: string, } catch {} } +/** + * A file modified within this long of the stat that read it gets its sync row + * stored with UNTRUSTED_SYNC_MTIME, which the fast path never matches. On a + * filesystem whose mtime granule is 1-2 s (HFS+, FAT, some network mounts), a + * same-size rewrite in that window would keep both mtime and size, and the + * fast path would skip it for good. The next update reads such a file once + * more and trusts it once its mtime is older: git's "racily clean" rule. + */ +const RACY_SYNC_WINDOW_MS = 2000; +const UNTRUSTED_SYNC_MTIME = -1; + /** * Maximum file size to index — prevents OOM on accidental binary inclusion. */ @@ -1780,6 +1791,7 @@ export async function reindexCollection( // Stat first — mtime+size fast-path (no read) let stat: ReturnType | null = null; + const statTimeMs = Date.now(); try { stat = statSync(filepath); } catch (err) { @@ -1797,6 +1809,7 @@ export async function reindexCollection( const mtimeMs = stat.mtimeMs; const size = stat.size; + const syncMtimeMs = statTimeMs - mtimeMs < RACY_SYNC_WINDOW_MS ? UNTRUSTED_SYNC_MTIME : mtimeMs; // Skip large files (>10MB) — prevents OOM if (size > REINDEX_MAX_FILE_SIZE) { @@ -1846,7 +1859,7 @@ export async function reindexCollection( // Hash matches cached sync state but mtime differed (clock skew, backup restore) — only update mtime cache if (cached && cached.content_hash === hash) { // Update sync state mtime/size only - upsertFileSyncState(db, collectionName, path, mtimeMs, size, hash, cached.document_id); + upsertFileSyncState(db, collectionName, path, syncMtimeMs, size, hash, cached.document_id); unchanged++; processed++; // Keep content in memory for metadata sync if needed? For speed, skip metadata sync on hash-match fast-path. @@ -1891,7 +1904,7 @@ export async function reindexCollection( } // Upsert sync state after successful indexing - upsertFileSyncState(db, collectionName, path, mtimeMs, size, hash, documentId); + upsertFileSyncState(db, collectionName, path, syncMtimeMs, size, hash, documentId); // Metadata extraction const extraction = syncDocumentMetadata(db, documentId, content, path, diff --git a/test/store.test.ts b/test/store.test.ts index eb6772c1d..a4f26da84 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3076,6 +3076,8 @@ describe("Reindex Collection file sync state (#962)", () => { const file = join(dir, "doc.md"); try { await writeFile(file, "# A\n\nalpha\n"); + // Older than the racy window, so the first pass stores a trusted row. + await utimes(file, new Date("2026-01-02T03:04:05Z"), new Date("2026-01-02T03:04:05Z")); await reindexCollection(store, dir, "**/*.md", "notes"); // Stat still works on an unreadable file; a read would fail with EACCES. await chmod(file, 0o000); @@ -3121,6 +3123,29 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("a same-size rewrite that keeps a recent mtime is still read", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-racy"); + const file = join(dir, "doc.md"); + try { + // A whole second ahead of the clock: inside the racy window however slow the run. + const mtime = new Date((Math.floor(Date.now() / 1000) + 1) * 1000); + await writeFile(file, "# A\n\nalpha\n"); + await utimes(file, mtime, mtime); + await reindexCollection(store, dir, "**/*.md", "notes"); + // Rewritten within the same mtime granule: same size, same mtime. + await writeFile(file, "# A\n\nbravo\n"); + await utimes(file, mtime, mtime); + + const result = await reindexCollection(store, dir, "**/*.md", "notes"); + expect(result).toMatchObject({ indexed: 0, updated: 1, unchanged: 0 }); + expect(activeBody(store, "notes", "doc.md")).toContain("bravo"); + } finally { + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("files over 10 MB are skipped with FILE_TOO_LARGE and not indexed", async () => { const store = await createTestStore(); const dir = await collectionDir("sync-too-large"); From e27f9f8e5b0cc78c917000ab7698238ff633c666 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 13:01:56 -0500 Subject: [PATCH 59/82] fix(search): break score ties by filepath when merging collections A keyword or vector search over several named collections merges one list per collection and sorts by score. The sort kept equal scores in the order the lists came, which is the order the collections were named, so the same content indexed in two collections came back from whichever was named first, and at the limit that decided which copy survived. Ties now go to the smaller filepath. --- src/store.ts | 8 +++++++- test/structured-search.test.ts | 14 ++++++++++++++ 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/src/store.ts b/src/store.ts index 0993191fb..c715bc469 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4439,11 +4439,17 @@ function mergeSearchResultsByScore(lists: SearchResult[][], limit: number): Sear if (!prev || r.score > prev.score) best.set(r.filepath, r); } } + // Ties go to the smaller filepath, so the order the collections were named + // never decides which of two equal hits survives the limit. return Array.from(best.values()) - .sort((a, b) => b.score - a.score) + .sort((a, b) => b.score - a.score || compareFilepaths(a, b)) .slice(0, limit); } +function compareFilepaths(a: { filepath: string }, b: { filepath: string }): number { + return a.filepath < b.filepath ? -1 : a.filepath > b.filepath ? 1 : 0; +} + export function searchFTS(db: Database, query: string, limit: number = 20, collectionName?: string | readonly string[], filter?: MetadataFilter): SearchResult[] { const names = scopedCollectionNames(collectionName); // Search each requested collection before merging/truncating so a large diff --git a/test/structured-search.test.ts b/test/structured-search.test.ts index 6d116e0f2..8f2ab08d6 100644 --- a/test/structured-search.test.ts +++ b/test/structured-search.test.ts @@ -18,6 +18,7 @@ import { hashContent, insertContent, insertDocument, + searchFTS, structuredSearch, validateSemanticQuery, validateLexQuery, @@ -369,6 +370,19 @@ describe("structuredSearch", () => { expect(results.map(r => r.displayPath)).toEqual(unscoped.map(r => r.displayPath)); } }); + + test("an exact tie between collections goes the same way whichever is named first", async () => { + const now = new Date().toISOString(); + const body = "quillmarrow notes"; + const hash = await hashContent(body); + insertContent(store.db, hash, body, now); + insertDocument(store.db, "alpha", "same.md", "same.md", hash, now, now); + insertDocument(store.db, "beta", "same.md", "same.md", hash, now, now); + + for (const collections of [["alpha", "beta"], ["beta", "alpha"]]) { + expect(searchFTS(store.db, "quillmarrow", 1, collections).map(r => r.filepath)).toEqual(["qmd://alpha/same.md"]); + } + }); }); // ============================================================================= From 1c061b056401753332ad41859dca63f98e6125fa Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 14:25:06 -0500 Subject: [PATCH 60/82] docs(changelog): credit #1000 by PR number qmd's release guidelines end each bullet from an external PR with `#NNN (thanks @username)`. #1000's bullet cited only its issue, #921; it now also ends with the PR and its author. --- CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 586afdb2a..f819cc4d8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -67,6 +67,7 @@ copy ran its own search. Cached expansions are deduplicated when read, so existing caches need no rebuild. Repeated lex lines used to count more than once in the rank fusion, so result order can shift slightly. (#921) + #1000 (thanks @ParkerRex) ## [2.8.3] - 2026-08-16 From 5c98af49e376afd8d1b82717b47ec708983118f7 Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:24:19 -0500 Subject: [PATCH 61/82] docs(changelog): record #946 and #1009 and credit #953 and #918 #946 and #1009 change structured-search rankings but add no changelog entry. The bullet says each structured line runs one search over the whole collection list, a keyword search for a lex line and a vector search for a vec or hyde line, so rankings no longer depend on the order collections are named, and credits both PRs. #953's bullet now ends with #953 and thanks #918's author for the approach it builds on, as qmd's release guidelines ask. --- CHANGELOG.md | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e9806e5f0..d668670c3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,7 +26,7 @@ `limit * 10` candidate window (#922). The scope is now applied to the full FTS5 match set, materialized once, so `search -c ` is exact for common terms; unscoped search keeps its early-terminating plan. Builds on - the approach in #918. + the approach in #918 (thanks @fxstein). #953 (thanks @Mr-Beasley) - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This @@ -76,6 +76,12 @@ existing caches need no rebuild. Repeated lex lines used to count more than once in the rank fusion, so result order can shift slightly. (#921) #1000 (thanks @ParkerRex) +- Structured searches over several collections (`qmd query` with + `lex:`/`vec:`/`hyde:` lines, the MCP `query` tool, SDK `queries`) run one + search per line over the whole collection list, a keyword search for a + `lex:` line and a vector search for a `vec:` or `hyde:` line, and fuse one + ranked list per search. Rankings no longer depend on the order the + collections are named. #946 (thanks @shalom-t), #1009 (thanks @xidus90) ## [2.8.3] - 2026-08-16 From 05cd983039177ff90cb32ee98b42a5fc27a00f82 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 19:35:39 -0500 Subject: [PATCH 62/82] perf(search): scope keyword search over several collections in one query searchFTS ran one FTS query per named collection and merged the lists by score (#775), so a scoped search over the default collections matched and ranked the whole index once per collection. That split existed because the scoped query only looked at a window of the best matches, where a large collection could crowd the others out. Since #918 and #953 the scope sees every match, so one query over the whole collection list returns the same results as the merge. searchFTS now filters by the whole list in one query and breaks score ties by filepath, as the merge did. On a copy of a large index, the 8 fixture queries over 13 default collections took 77.6 s the old way and 6.1 s in one query, with the same top 20 for each. --- CHANGELOG.md | 7 +++++-- src/store.ts | 22 ++++++++++------------ test/store.test.ts | 37 +++++++++++++++++++++++++++++++++++++ 3 files changed, 52 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d668670c3..34488839f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,8 +25,11 @@ incomplete results when stronger matches outside the scope fill the old `limit * 10` candidate window (#922). The scope is now applied to the full FTS5 match set, materialized once, so `search -c ` is exact for - common terms; unscoped search keeps its early-terminating plan. Builds on - the approach in #918 (thanks @fxstein). #953 (thanks @Mr-Beasley) + common terms; unscoped search keeps its early-terminating plan. A search over + several collections, such as the default ones, runs one keyword query instead + of one per collection, which made keyword search about 12x faster over 13 + collections (thanks @brettdavies). Builds on the approach in #918 (thanks + @fxstein). #953 (thanks @Mr-Beasley) - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This diff --git a/src/store.ts b/src/store.ts index c715bc469..6cdada20f 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4452,12 +4452,6 @@ function compareFilepaths(a: { filepath: string }, b: { filepath: string }): num export function searchFTS(db: Database, query: string, limit: number = 20, collectionName?: string | readonly string[], filter?: MetadataFilter): SearchResult[] { const names = scopedCollectionNames(collectionName); - // Search each requested collection before merging/truncating so a large - // unrelated collection cannot occupy global top-k and starve the rest (#775). - if (names && names.length > 1) { - return mergeSearchResultsByScore(names.map(name => searchFTS(db, query, limit, name, filter)), limit); - } - const collectionFilter = names?.[0]; const ftsQuery = buildFTS5Query(query); if (!ftsQuery) return []; @@ -4476,7 +4470,10 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle // the window (#922). MATERIALIZED keeps the planner from flattening the CTE // and folding the filter back into the MATCH. The set is corpus-bounded: // at most one row per matching document. - const scoped = Boolean(collectionFilter || filter); + // Because the scope sees every match, one query over the whole collection + // list returns what a query per collection merged by score would (#775): + // a large collection can no longer crowd the others out of a window. + const scoped = Boolean(names || filter); let sql = ` WITH fts_matches AS ${scoped ? "MATERIALIZED " : ""}( @@ -4500,9 +4497,9 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle WHERE d.active = 1 `; - if (collectionFilter) { - sql += ` AND d.collection = ?`; - params.push(String(collectionFilter)); + if (names) { + sql += ` AND d.collection IN (SELECT value FROM json_each(?))`; + params.push(JSON.stringify(names)); } if (filter) { @@ -4513,8 +4510,9 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle params.push(...compiledFilter.params); } - // bm25 lower is better; sort ascending. - sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`; + // bm25 lower is better; sort ascending, ties by filepath as in + // mergeSearchResultsByScore. + sql += ` ORDER BY fm.bm25_score ASC, filepath ASC LIMIT ?`; params.push(limit); const rows = db.prepare(sql).all(...params) as { filepath: string; display_path: string; title: string; body: string; hash: string; bm25_score: number; metadata_json: string | null }[]; diff --git a/test/store.test.ts b/test/store.test.ts index 7c83d58ec..6f9b63578 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3683,6 +3683,43 @@ describe("Collection-scoped keyword search", () => { } }); + test("searchFTS over several collections runs one keyword query and returns the same results", async () => { + const store = await createTestStore(); + try { + const { smallA, smallB } = await crowdedCollections(store, 40); + const scope = ["crowd-large", smallA, smallB]; + const perCollection = scope.flatMap(name => store.searchFTS("zebra", 5, name)) + .sort((a, b) => b.score - a.score || (a.filepath < b.filepath ? -1 : a.filepath > b.filepath ? 1 : 0)) + .slice(0, 5); + // Counts the keyword queries the scoped search runs. + let ftsQueries = 0; + const counting: Database = { + prepare: (sql: string) => { + const real = store.db.prepare(sql); + if (!sql.includes("documents_fts MATCH")) return real; + return { + ...real, + run: real.run.bind(real), + get: real.get.bind(real), + iterate: real.iterate.bind(real), + all: (...params: Parameters) => { ftsQueries++; return real.all(...params); }, + }; + }, + transaction: (fn) => store.db.transaction(fn), + exec: (sql: string) => store.db.exec(sql), + loadExtension: (path: string) => store.db.loadExtension(path), + close: () => store.db.close(), + }; + + const { searchFTS } = await import("../src/store.js"); + const results = searchFTS(counting, "zebra", 5, scope); + expect(ftsQueries).toBe(1); + expect(results.map(r => r.filepath)).toEqual(perCollection.map(r => r.filepath)); + } finally { + await cleanupTestDb(store); + } + }); + test("structuredSearch over two crowded-out collections returns a match from each", async () => { const store = await createTestStore(); try { From eb0b105c299b185d60adee5e4da3ff11dd6f927a Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:24:50 -0500 Subject: [PATCH 63/82] fix(store): keep the body cap in partitioned vector search #983 moved searchVec's body read into its own lookup, which fetched the whole document again and dropped #962's cap. The lookup now returns the first 262,144 characters, like searchFTS, and the three reads that cap a body (searchFTS, this lookup and getHashesForEmbedding) share BODY_CAP_CHARS. #983's race test intercepts the lookup's SQL, so its expected string follows. The #962 body-cap test "searchVec returns at most 256 KiB of a document body" passes again. --- src/store.ts | 10 ++++++---- test/store.test.ts | 2 +- 2 files changed, 7 insertions(+), 5 deletions(-) diff --git a/src/store.ts b/src/store.ts index 4a031ecc4..19e997d3e 100644 --- a/src/store.ts +++ b/src/store.ts @@ -138,9 +138,9 @@ export function splitGlobMask(mask: string): string[] { export const DEFAULT_MULTI_GET_MAX_BYTES = 64 * 1024; // 64KB /** - * Characters of a document body that search results and getHashesForEmbedding - * return, so one very large document cannot put its whole text on the heap - * per result. + * Characters of a document body that search results, the vector body lookup + * and getHashesForEmbedding return (SQLite's substr counts characters), so one + * very large document cannot put its whole text on the heap per result. */ const BODY_CAP_CHARS = 262_144; @@ -4822,7 +4822,9 @@ export async function searchVec(db: Database, query: string, model: string, limi const scan = knnVecScanner(db, collectionIds !== undefined, eligible !== undefined); const resolve = vecDocumentResolver(db, filter); const queryVec = new Float32Array(embedding); - const bodyOf = db.prepare(`SELECT doc FROM content WHERE hash = ?`); + // Bodies are capped at BODY_CAP_CHARS, as in searchFTS, so a large document cannot + // put its whole text on the heap for each result. + const bodyOf = db.prepare(`SELECT ${cappedBodySql("doc")} AS doc FROM content WHERE hash = ?`); // Each target yields its own nearest `limit` documents (or all it holds), so // merging them by distance gives the scope's exact nearest `limit`. diff --git a/test/store.test.ts b/test/store.test.ts index ca8cb07be..76160d5c0 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -4925,7 +4925,7 @@ describe("Vector Search collection filter", () => { // Replays another process's orphaned-content cleanup landing between // document resolution and the body read. - const bodySql = "SELECT doc FROM content WHERE hash = ?"; + const bodySql = "SELECT CASE WHEN length(CAST(doc AS BLOB)) <= 262144 THEN doc ELSE substr(doc, 1, 262144) END AS doc FROM content WHERE hash = ?"; const racing: Database = { prepare: (sql: string) => { const statement = store.db.prepare(sql); From abf570f068e016d4e42a7c891df49d965fc02ba4 Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 11:04:54 -0500 Subject: [PATCH 64/82] test(store): point #1000's cached-expansion test at the partitioned table #1000's test stubs a legacy vectors_vec table to switch on the vector branch of hybridQuery. With #983, hasVectorIndex() only recognizes the partitioned table, so the branch never ran and the test failed with "expected spy to be called 1 times, but got 0 times". The stub now creates the partitioned table, as #983 already does for the tests next to it. The assertions are unchanged. --- test/store.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/store.test.ts b/test/store.test.ts index 76160d5c0..b06598954 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -1414,7 +1414,7 @@ describe("Query expansion cache (#818)", () => { const store = await createTestStore(); const embedModel = "hf:ggml-org/embeddinggemma-300M-GGUF/embeddinggemma-300M-Q8_0.gguf"; const embedBatchSpy = vi.fn(async (texts: string[]) => texts.map(() => ({ embedding: [1, 2, 3], model: embedModel }))); - store.db.exec(`CREATE TABLE vectors_vec (hash_seq TEXT PRIMARY KEY, embedding BLOB)`); + store.db.exec(`CREATE TABLE ${VEC_TABLE} (collection_id INTEGER, embedding BLOB)`); store.llm = { embedModelName: embedModel, embedBatch: embedBatchSpy } as any; store.searchVec = vi.fn(async () => [] as SearchResult[]) as any; try { From 84d39e0e9804d9de0c9a7bf42e2b2204d4271b78 Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:24:58 -0500 Subject: [PATCH 65/82] test(store): check #983's atomic rename keeps the fast path Renaming a collection moves its #962 sync rows, and #983's atomic rename carries that move inside its transaction. The test renames a collection through #983's `renameCollection`, checks the sync row is under the new name at once, and makes the file unreadable to show the first pass under the new name reads nothing. Its file is backdated past the racy-mtime window so the first pass stores a trusted row. --- test/store.test.ts | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/test/store.test.ts b/test/store.test.ts index b06598954..efbd33e26 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3603,6 +3603,30 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("after #983's atomic collection rename the new name takes the fast path at once", async () => { + const store = await createTestStore(); + const dir = await collectionDir("sync-atomic-rename"); + const file = join(dir, "doc.md"); + try { + await writeFile(file, "# A\n\nalpha\n"); + await utimes(file, new Date("2026-01-02T03:04:05Z"), new Date("2026-01-02T03:04:05Z")); + await reindexCollection(store, dir, "**/*.md", "notes"); + renameCollection(store.db, "notes", "archive"); + expect(syncRowCount(store, "archive", "doc.md")).toBe(1); + + // The first pass under the new name would report EACCES if it read the file. + await chmod(file, 0o000); + const first = await reindexCollection(store, dir, "**/*.md", "archive"); + expect(first.skippedFiles).toEqual([]); + expect(first).toMatchObject({ indexed: 0, updated: 0, unchanged: 1, removed: 0 }); + expect(activeBody(store, "archive", "doc.md")).toContain("alpha"); + } finally { + await chmod(file, 0o644); + await rm(dir, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("searchFTS returns at most 256 KiB of a document body", async () => { const store = await createTestStore(); try { From 7c7997f7e981e05183c5237ed40ad25ab0db159f Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 12:39:28 -0500 Subject: [PATCH 66/82] fix(search): restrict a filtered vector scan with a subquery, not a list #983 restricts a filtered vector scan to the rows the metadata filter admits by reading them with all() and binding them as one JSON list, so a broad filter on a large index put every eligible row on the heap once per vector search. #962's iterate() change had bounded that read at 20,000 rows. The scan's `rowid IN` restriction is now the eligibility query itself, which vec0 applies inside the scan as it did the list. Results stay exact however many rows the filter admits, and none of those rows reaches the heap. searchVec still skips a scoped collection with no eligible row, now from a DISTINCT query over collection ids. --- src/store.ts | 75 ++++++++++++++++++------------------ test/metadata-search.test.ts | 42 ++++++++++++++++++++ 2 files changed, 79 insertions(+), 38 deletions(-) diff --git a/src/store.ts b/src/store.ts index 19e997d3e..2f4ad0652 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4639,10 +4639,9 @@ interface VecMatch { distance: number; } -/** One KNN scan target: a partition (or the whole table) and the rows a filter admits in it. */ +/** One KNN scan target: a collection's partition, or the whole table when no scope is given. */ interface VecScanTarget { collectionId?: number; - eligibleRowids?: readonly number[]; } /** The document behind a vector match, at its nearest chunk. */ @@ -4676,12 +4675,15 @@ function compileCurrentMetadataFilter(filter: MetadataFilter): { sql: string; pa * whole index and is never crowded out by a larger one (#775, #791, #803). A * `rowid IN` restriction is applied inside the scan the same way, so a * selective metadata filter gets an exact top-k of its own rows instead of - * whatever survives a post-filter of a larger top-k. + * whatever survives a post-filter of a larger top-k. The restriction is the + * eligibility subquery itself rather than a bound list of rowids, so however + * many rows a filter admits, none of them is read onto the heap. */ -function knnVecScanner(db: Database, partitioned: boolean, restricted: boolean): (embedding: Float32Array, k: number, target: VecScanTarget) => VecMatch[] { +function knnVecScanner(db: Database, partitioned: boolean, filter?: MetadataFilter): (embedding: Float32Array, k: number, target: VecScanTarget) => VecMatch[] { const conditions = ["embedding MATCH ?", "k = ?"]; if (partitioned) conditions.push("collection_id = ?"); - if (restricted) conditions.push("rowid IN (SELECT value FROM json_each(?))"); + const eligible = filter ? metadataEligibleRowsSql(filter) : undefined; + if (eligible) conditions.push(`rowid IN (SELECT vr.id ${eligible.sql}${partitioned ? " AND vr.collection_id = ?" : ""})`); const statement = db.prepare(` SELECT rowid, distance FROM ${VEC_TABLE} @@ -4689,40 +4691,39 @@ function knnVecScanner(db: Database, partitioned: boolean, restricted: boolean): `); return (embedding, k, target) => { const params: SQLiteValue[] = [embedding, k]; - if (partitioned) params.push(vecInteger(target.collectionId ?? 0)); - if (restricted) params.push(rowidList(target.eligibleRowids ?? [])); + const partition = vecInteger(target.collectionId ?? 0); + if (partitioned) params.push(partition); + if (eligible) { + params.push(...eligible.params); + if (partitioned) params.push(partition); + } return statement.all(...params) as VecMatch[]; }; } /** - * Vector rows of active documents a metadata filter admits, keyed by - * collection id. A row is eligible when any document of its collection that - * holds the row's content passes the filter. + * FROM and WHERE clauses selecting, as `vr`, the vector rows of active + * documents a metadata filter admits. A row is eligible when any document of + * its collection that holds the row's content passes the filter. */ -function metadataEligibleVectorRows(db: Database, filter: MetadataFilter, collectionIds?: readonly number[]): Map { +function metadataEligibleRowsSql(filter: MetadataFilter): { sql: string; params: SQLiteValue[] } { const current = compileCurrentMetadataFilter(filter); - const params: SQLiteValue[] = [...current.params]; - let scope = ""; - if (collectionIds) { - scope = ` AND vr.collection_id IN (SELECT value FROM json_each(?))`; - params.push(JSON.stringify(collectionIds)); - } - const rows = db.prepare(` - SELECT DISTINCT vr.id AS rowid, vr.collection_id AS collectionId - FROM documents d - JOIN document_metadata dm ON dm.document_id = d.id - JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.name = d.collection - JOIN ${VEC_ROWS_TABLE} vr ON vr.hash = d.hash AND vr.collection_id = ci.id - WHERE d.active = 1 AND ${current.sql}${scope} - `).all(...params) as { rowid: number; collectionId: number }[]; - const byCollection = new Map(); - for (const row of rows) { - const list = byCollection.get(row.collectionId); - if (list) list.push(row.rowid); - else byCollection.set(row.collectionId, [row.rowid]); - } - return byCollection; + return { + sql: ` + FROM documents d + JOIN document_metadata dm ON dm.document_id = d.id + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.name = d.collection + JOIN ${VEC_ROWS_TABLE} vr ON vr.hash = d.hash AND vr.collection_id = ci.id + WHERE d.active = 1 AND ${current.sql}`, + params: current.params, + }; +} + +/** Ids of the collections holding at least one vector row a metadata filter admits. */ +function metadataEligibleCollections(db: Database, filter: MetadataFilter): Set { + const eligible = metadataEligibleRowsSql(filter); + const rows = db.prepare(`SELECT DISTINCT vr.collection_id AS collectionId ${eligible.sql}`).all(...eligible.params) as { collectionId: number }[]; + return new Set(rows.map(row => row.collectionId)); } /** @@ -4802,7 +4803,7 @@ export async function searchVec(db: Database, query: string, model: string, limi collectionIds = Array.from(resolveCollectionIds(db, names).values()); if (collectionIds.length === 0) return []; } - const eligible = filter ? metadataEligibleVectorRows(db, filter, collectionIds) : undefined; + const eligible = filter ? metadataEligibleCollections(db, filter) : undefined; // IMPORTANT: We use a two-step query approach here because sqlite-vec virtual tables // hang indefinitely when combined with JOINs in the same query. Do NOT try to @@ -4814,12 +4815,10 @@ export async function searchVec(db: Database, query: string, model: string, limi // member rather than `collection_id IN (...)`: the IN form yields k rows per // value only because SQLite runs vec0's filter once per value, which is a // planner detail rather than a vec0 contract. - const targets: VecScanTarget[] = collectionIds - ? collectionIds.map(collectionId => ({ collectionId, eligibleRowids: eligible?.get(collectionId) })) - : [{ eligibleRowids: eligible && Array.from(eligible.values()).flat() }]; - const scanTargets = eligible ? targets.filter(t => t.eligibleRowids?.length) : targets; + const scanned = collectionIds && eligible ? collectionIds.filter(id => eligible.has(id)) : collectionIds; + const scanTargets: VecScanTarget[] = scanned ? scanned.map(collectionId => ({ collectionId })) : eligible?.size === 0 ? [] : [{}]; if (scanTargets.length === 0) return []; - const scan = knnVecScanner(db, collectionIds !== undefined, eligible !== undefined); + const scan = knnVecScanner(db, collectionIds !== undefined, filter); const resolve = vecDocumentResolver(db, filter); const queryVec = new Float32Array(embedding); // Bodies are capped at BODY_CAP_CHARS, as in searchFTS, so a large document cannot diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index 9c0595628..c17731999 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -23,6 +23,7 @@ import { import { replaceDocumentMetadata, syncDocumentMetadata } from "../src/metadata-store.js"; import { METADATA_EXTRACTION_VERSION, type DocumentMetadata } from "../src/metadata.js"; import { parseMetadataFilter, type MetadataFilter } from "../src/metadata-filter.js"; +import type { Database, SQLiteValue } from "../src/db.js"; let testDir: string; let store: Store; @@ -335,6 +336,47 @@ describe("searchVec with metadata filter", () => { expect(filtered.every(r => r.metadata.eligible === true)).toBe(true); }, 120_000); + test("a filtered vector scan binds no list of eligible rows, so none is held on the heap", async () => { + store.ensureVecTable(3); + // One eligible document of 2,000 chunks, and closer ineligible documents. + const { hash } = await insertDoc("book", "long-eligible.md", "# Long eligible", { eligible: true }); + const now = new Date().toISOString(); + store.db.transaction(() => { + for (let seq = 0; seq < 2_000; seq++) insertEmbedding(store.db, hash, seq, seq, new Float32Array([0, 1, 0]), model, now, 2_000); + })(); + for (let i = 0; i < 50; i++) { + await insertEmbeddedDoc("book", `closer-${i}.md`, `# Closer ${i}`, [1, 0, 0], { eligible: false }); + } + // Every string bound to a vector scan: a JSON list of eligible rowids would show here. + const bound: string[] = []; + const recording: Database = { + prepare: (sql: string) => { + const real = store.db.prepare(sql); + if (!sql.includes("MATCH")) return real; + return { + ...real, + run: real.run.bind(real), + get: real.get.bind(real), + iterate: real.iterate.bind(real), + all: (...params: SQLiteValue[]) => { + for (const param of params) if (typeof param === "string" && param.startsWith("[")) bound.push(param); + return real.all(...params); + }, + }; + }, + transaction: (fn) => store.db.transaction(fn), + exec: (sql: string) => store.db.exec(sql), + loadExtension: (path: string) => store.db.loadExtension(path), + close: () => store.db.close(), + }; + + for (const scope of ["book", undefined]) { + const results = await searchVec(recording, "q", model, 5, scope, undefined, queryEmbedding, undefined, eligibleOnly); + expect(results.map(r => r.displayPath)).toEqual(["book/long-eligible.md"]); + } + expect(bound).toEqual([]); + }); + test("shared content hash within one collection returns only the matching document path", async () => { store.ensureVecTable(3); const body = "# Shared body"; From 0f89d9be4b092aec116cd464f2ec401dea6c037d Mon Sep 17 00:00:00 2001 From: Brett Date: Tue, 29 Sep 2026 13:02:50 -0500 Subject: [PATCH 67/82] fix(search): break vector distance ties by filepath across partitions #983's searchVec runs one KNN per collection in scope and merges the results by distance. The sort kept equal distances in the order the collections were named, so the same content in two collections came back from whichever was named first. Ties now go to the smaller filepath, as they do in the keyword merge. --- src/store.ts | 5 +++-- test/store.test.ts | 17 +++++++++++++++++ 2 files changed, 20 insertions(+), 2 deletions(-) diff --git a/src/store.ts b/src/store.ts index 2f4ad0652..b3b81d66f 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4826,10 +4826,11 @@ export async function searchVec(db: Database, query: string, model: string, limi const bodyOf = db.prepare(`SELECT ${cappedBodySql("doc")} AS doc FROM content WHERE hash = ?`); // Each target yields its own nearest `limit` documents (or all it holds), so - // merging them by distance gives the scope's exact nearest `limit`. + // merging them by distance gives the scope's exact nearest `limit`. Ties go + // to the smaller filepath, as in mergeSearchResultsByScore. return scanTargets .flatMap(target => nearestVecDocuments(scan, resolve, queryVec, limit, target)) - .sort((a, b) => a.distance - b.distance) + .sort((a, b) => a.distance - b.distance || compareFilepaths(a, b)) .slice(0, limit) .flatMap((row): SearchResult[] => { // The body is read after resolution, outside its snapshot: another diff --git a/test/store.test.ts b/test/store.test.ts index efbd33e26..7a8618db9 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -4391,6 +4391,23 @@ describe("Integration", () => { // ============================================================================= describe("Vector Search collection filter", () => { + test("an exact vector tie between collections goes the same way whichever is named first", async () => { + const store = await createTestStore(); + const alpha = await createTestCollection({ name: "alpha", pwd: "/test/alpha" }); + const beta = await createTestCollection({ name: "beta", pwd: "/test/beta" }); + store.ensureVecTable(3); + for (const collection of [alpha, beta]) { + await insertTestDocument(store.db, collection, { name: "same", hash: "samehash", body: "Same body", displayPath: "same.md" }); + } + store.insertEmbedding("samehash", 0, 0, new Float32Array([1, 0, 0]), "test", new Date().toISOString()); + + for (const collections of [["alpha", "beta"], ["beta", "alpha"]]) { + const results = await store.searchVec("ignored", "test-model", 1, collections, undefined, [1, 0, 0]); + expect(results.map(r => r.filepath)).toEqual(["qmd://alpha/same.md"]); + } + await cleanupTestDb(store); + }); + test("searchVec finds docs in a small collection crowded by a large one (#791, #803)", async () => { const store = await createTestStore(); const large = await createTestCollection({ name: "large", pwd: "/test/large" }); From 20fa36e42bd0f0aa870b9d2aafb1107a51e03cdd Mon Sep 17 00:00:00 2001 From: Parker Rex Date: Sat, 26 Sep 2026 09:49:02 -0500 Subject: [PATCH 68/82] fix(update): leave the index unchanged when a collection root is missing A collection whose root folder is absent at update time (unmounted drive, offline share, renamed folder) globbed to an empty file list. The deactivation pass then read that as "every file was deleted", search went empty, and a `qmd cleanup` before the folder returned hard-deleted the rows. Check that the root is a directory before globbing. If it is not, return without touching the index and report the root as ROOT_MISSING so the CLI can warn. A genuinely empty folder still deactivates everything. Fixes #989 Co-Authored-By: Claude Fable 5.1 (cherry picked from commit f8416e7bf9a7364f26822d45b6681b740b3b1582) --- CHANGELOG.md | 7 +++++++ src/cli/qmd.ts | 7 +++++-- src/store.ts | 20 ++++++++++++++++++++ test/cli.test.ts | 21 +++++++++++++++++++++ test/store.test.ts | 36 ++++++++++++++++++++++++++++++++++++ 5 files changed, 89 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7e742ee83..57c1fa124 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -31,6 +31,13 @@ collections (thanks @brettdavies). Builds on the approach in #918 (thanks @fxstein). #953 (thanks @Mr-Beasley) +- `qmd update` no longer deactivates every document in a collection whose root + folder is missing, for example an unmounted drive or an offline network share. + The folder globbed to an empty list, which read as "every file was deleted", + so search went empty and a `qmd cleanup` before the drive came back deleted + the rows for good. The collection is now reported as not found and its index + is left unchanged. Permission and I/O errors still fail with their original + cause. (#989) - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This keeps chunk boundaries aligned with the model that creates and verifies the diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index f1297bf03..e6e834ebb 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -2287,7 +2287,9 @@ function reportSkippedReads(skippedFiles: { file: string; code: string }[]): voi if (skippedFiles.length === 0) return; const sizeLimitMb = Math.round(REINDEX_MAX_FILE_SIZE / (1024 * 1024)); for (const skipped of skippedFiles) { - if (skipped.code === "OUTSIDE_COLLECTION") { + if (skipped.code === "ROOT_MISSING") { + console.warn(`⚠ Collection root not found, index left unchanged: ${skipped.file}`); + } else if (skipped.code === "OUTSIDE_COLLECTION") { console.warn(`⚠ Skipped file outside collection: ${skipped.file}`); } else if (skipped.code === "FILE_TOO_LARGE") { console.warn(`⚠ Skipped file over ${sizeLimitMb} MB: ${skipped.file}`); @@ -2297,7 +2299,8 @@ function reportSkippedReads(skippedFiles: { file: string; code: string }[]): voi } const escaped = skippedFiles.filter(f => f.code === "OUTSIDE_COLLECTION").length; const tooLarge = skippedFiles.filter(f => f.code === "FILE_TOO_LARGE").length; - const unreadable = skippedFiles.length - escaped - tooLarge; + const rootMissing = skippedFiles.filter(f => f.code === "ROOT_MISSING").length; + const unreadable = skippedFiles.length - escaped - tooLarge - rootMissing; if (escaped) console.warn(`Skipped ${escaped} file(s) outside the collection root`); if (tooLarge) console.warn(`Skipped ${tooLarge} file(s) over ${sizeLimitMb} MB`); if (unreadable) console.warn(`Skipped ${unreadable} unreadable file(s)`); diff --git a/src/store.ts b/src/store.ts index b3b81d66f..92f273029 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1880,6 +1880,26 @@ async function reindexCollectionIn( ...excludeDirs.map(d => `**/${d}/**`), ...(options?.ignorePatterns || []), ]; + + // A missing root (unmounted drive, offline share, deleted folder) globs to + // an empty list, which the deactivation pass below would read as "every + // file was deleted". Leave the index alone and report the root instead. + let rootIsDirectory = false; + try { + rootIsDirectory = statSync(collectionPath).isDirectory(); + } catch (error) { + const code = fsErrorCode(error); + if (code !== "ENOENT" && code !== "ENOTDIR") throw error; + } + if (!rootIsDirectory) { + return { + indexed: 0, updated: 0, unchanged: 0, removed: 0, orphanedCleaned: 0, + skipped: 1, + skippedFiles: [{ file: collectionPath, code: "ROOT_MISSING" }], + metadataErrors: 0, + }; + } + const allFiles: string[] = await fastGlob(splitGlobMask(globPattern), { cwd: collectionPath, onlyFiles: true, diff --git a/test/cli.test.ts b/test/cli.test.ts index bb23030d6..de4473b73 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -1267,6 +1267,27 @@ describe("CLI Multi-Get Command", () => { }); describe("CLI Update Command", () => { + test.skipIf(process.platform === "win32" || process.getuid?.() === 0)("reports collection root permission failures and exits non-zero", async () => { + const env = await createIsolatedTestEnv("update-root-permissions"); + const parent = join(testDir, "restricted-root"); + const collectionPath = join(parent, "docs"); + await mkdir(collectionPath, { recursive: true }); + await writeFile(join(collectionPath, "readme.md"), "# Readme\n\nPreserve this indexed document.\n"); + const added = await runQmd(["collection", "add", collectionPath, "--name", "docs"], env); + expect(added.exitCode).toBe(0); + await chmod(parent, 0o000); + try { + const result = await runQmd(["update"], env); + expect(result.exitCode).toBe(1); + expect(result.stderr).toContain("EACCES"); + const search = await runQmd(["search", "Preserve", "--json"], env); + expect(search.exitCode).toBe(0); + expect(search.stdout).toContain("readme.md"); + } finally { + await chmod(parent, 0o700); + } + }, 60000); + let localDbPath: string; beforeEach(async () => { diff --git a/test/store.test.ts b/test/store.test.ts index 7a8618db9..f2a160f72 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3027,6 +3027,42 @@ describe("Reindex Collection", () => { expect(paths.map(r => r.path)).toEqual(["X - b.md", "a.md"]); }); + test("leaves the index unchanged when the collection root is missing", async () => { + const store = await createTestStore(); + const collectionName = "unmounted"; + const parent = join(testDir, `unmounted-${Date.now()}-${Math.random().toString(36).slice(2)}`); + const collectionPath = join(parent, "notes"); + await mkdir(collectionPath, { recursive: true }); + await writeFile(join(collectionPath, "a.md"), "# A\n\nalpha\n"); + await writeFile(join(collectionPath, "b.md"), "# B\n\nbravo\n"); + + try { + const initial = await reindexCollection(store, collectionPath, "**/*.md", collectionName); + expect(initial.indexed).toBe(2); + + // An unmounted drive or offline share looks exactly like an empty folder to + // the glob. That must not read as "every file was deleted". + await rename(collectionPath, join(parent, "notes.unmounted")); + const whileMissing = await reindexCollection(store, collectionPath, "**/*.md", collectionName); + expect(whileMissing.removed).toBe(0); + expect(whileMissing.skipped).toBe(1); + expect(whileMissing.skippedFiles[0]!.code).toBe("ROOT_MISSING"); + + const active = store.db.prepare(` + SELECT COUNT(*) AS count FROM documents WHERE collection = ? AND active = 1 + `).get(collectionName) as { count: number }; + expect(active.count).toBe(2); + + // A genuinely empty folder still deactivates everything. + await mkdir(collectionPath); + const whileEmpty = await reindexCollection(store, collectionPath, "**/*.md", collectionName); + expect(whileEmpty.removed).toBe(2); + } finally { + await rm(parent, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("does not index a file symlink whose target is outside the collection", async () => { const store = await createTestStore(); const parent = join(testDir, `escape-sym-${Date.now()}-${Math.random().toString(36).slice(2)}`); From e999c61a1c4a63ba2343334f4ede48f97a65f8db Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:23:46 -0500 Subject: [PATCH 69/82] docs(changelog): record #815's cache and credit #923, #978 and #995 #815 changes what users see (the MCP server's instructions are cached per store for up to 60 seconds) but adds no changelog entry; this adds one, credited to #815. The bullets #923 and #978 brought had no credit, and #995's credited its issue (#994) rather than the PR, so each now ends with its PR number and author, as qmd's release guidelines ask. --- CHANGELOG.md | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4440b1343..1baaaacd9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,8 +6,10 @@ - `qmd doctor` selects vector sample identities before loading document bodies, avoiding excessive SQLite memory use on large indexes with duplicate paths. + #978 (thanks @naveenspark) - Vector diagnostics match passages by their saved character position, avoiding false mismatches when earlier chunks change the sequence numbering. + #978 (thanks @naveenspark) ### Added @@ -28,12 +30,17 @@ materialized the full body once per legacy chunk and per active path before `LIMIT 1` discarded it, the same pattern as the doctor vector-sample check (#978). It now picks the sample row through indexes and loads only that - row's body; the sampled chunk is unchanged. #994 (thanks @mjaverto) + row's body; the sampled chunk is unchanged (#994). #995 (thanks @mjaverto) ### Changed +- The MCP server caches its instructions per store for up to 60 seconds instead + of rebuilding them, with a full index-status scan, for every HTTP request. + Concurrent requests share one build and a failed build is not cached; index + changes from other processes show up in the instructions within a minute. + #815 (thanks @fxstein) - Avoid a redundant runtime process on packaged CLI calls where the launcher is - already running under its selected Node or Bun runtime. + already running under its selected Node or Bun runtime. #923 (thanks @ilepn) ## [2.8.3] - 2026-08-16 From f277cc9386f62a98aa2158efe5a84891569fb61a Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:24:00 -0500 Subject: [PATCH 70/82] docs(changelog): record #962's re-index fast path and body cap #962 changes what users see but adds no changelog entry. Two bullets describe the behavior as it stands with the fixes above: `qmd update` skips files whose mtime and size are unchanged, a collection removed and added back is re-indexed while a renamed one keeps its entries, stale metadata is re-extracted, `qmd update` and `qmd collection add` skip files over 10 MB, and search results carry at most the first 262,144 characters of a body (SQLite's substr counts characters), which also bounds the passage the reranker scores. Both credit #962. --- CHANGELOG.md | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1baaaacd9..3727808c3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -41,6 +41,21 @@ #815 (thanks @fxstein) - Avoid a redundant runtime process on packaged CLI calls where the launcher is already running under its selected Node or Bun runtime. #923 (thanks @ilepn) +- `qmd update` no longer reads files whose modification time and size match + the last pass (tracked in a new `file_sync_state` table), so re-indexing an + unchanged collection takes seconds. A cached entry counts only while its + document is still active at that path with that content, so a collection + removed and added back is re-indexed, while a renamed one keeps its entries; + files with missing or outdated metadata are still re-extracted. `qmd update` + and `qmd collection add` skip files over 10 MB with `FILE_TOO_LARGE`, and + `qmd update` deactivates a previously indexed file that becomes empty or is + over 10 MB, including one indexed by an earlier release. #962 (thanks + @rikvanriel) +- Search results carry at most the first 262,144 characters of each document + body (`qmd search --full`, MCP results, keyword and vector hits), which + bounds memory on indexes with very large documents. A match past that point + is still found, but its snippet, and the passage the reranker scores, come + from the start of the document. #962 (thanks @rikvanriel) ## [2.8.3] - 2026-08-16 From 5ed47afa94aa46078c21a276662fe2f5a3109f8a Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:25:43 -0500 Subject: [PATCH 71/82] feat(cleanup): repack the partitioned vector table within each partition The repack reads the partitioned table's shadow tables through the layout resolver, skips the newest chunk of every partition rather than one global newest chunk (vec0 appends each partition's inserts to its own tail), and re-inserts each moved row under its existing rowid and partition so the `vector_rows` mapping stays valid. Each chunk is read again inside its transaction, so a row deleted since the plan, whose rowid another collection has taken over, is not moved into this partition. The trigger counts occupancy per partition, since rows of different partitions never share a chunk: a packed table needs ceil(rows / chunk size) chunks in each partition, and a repack also needs at least one chunk it would move. The `qmd cleanup --dry-run` projection counts partition rows whose (hash, collection) has no active document and leaves out a chunk whose rows are all orphans, as vec0 drops it with its last row, so the preview predicts the same decision. The test that kept a hash-keyed legacy table untouched goes with the code: a legacy table no longer survives a store open. Tests cover rows staying in their partition, a packed multi-partition table not repacking again, a table below the trigger with no chunk to move, the orphan-only chunk in a dry run, and a rowid reused by another partition mid-repack. The changelog credits #937. --- CHANGELOG.md | 1 + src/store.ts | 171 ++++++++++++++++++++++++++----------------- test/cleanup.test.ts | 120 +++++++++++++++++++++++------- 3 files changed, 197 insertions(+), 95 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ed15f73b3..a89c471d5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -125,6 +125,7 @@ to the tail, one chunk per short transaction, so it never holds the write lock for long and an interrupted run leaves a consistent table. The cleanup output and `qmd cleanup --dry-run` report the chunk counts. + #937 (thanks @brettdavies) ## [2.8.3] - 2026-08-16 diff --git a/src/store.ts b/src/store.ts index 6d1149f58..4b3b0f6dd 100644 --- a/src/store.ts +++ b/src/store.ts @@ -37,6 +37,7 @@ import { vecLayout, vecTableReadable, type MissingPartitionRow, + type VecLayout, } from "./vec-layout.js"; import { FTS_SYNC_TRIGGERS_VERSION, @@ -3283,48 +3284,80 @@ export type VectorTableLayout = { interface VecChunkRow { chunkId: number; size: number; + partition: number; validity: Uint8Array; rowids: Uint8Array; } -function vecTableReadable(db: Database): boolean { - if (!isSqliteVecAvailable()) return false; - try { - db.prepare(`SELECT 1 FROM vectors_vec LIMIT 0`).get(); - return true; - } catch { - return false; - } +/** A plan for repackVectors: the table's chunk layout and the chunks it would move. */ +type VecRepackPlan = { layout: VectorTableLayout; sparse: number[] }; + +/** The partitioned vec0 table when sqlite-vec is loaded and the table answers; null otherwise. */ +function readableVectorLayout(db: Database): Extract | null { + if (!isSqliteVecAvailable()) return null; + const layout = vecLayout(db); + return layout.kind === "partitioned" && vecTableReadable(db, layout) ? layout : null; } /** - * Chunk layout of vectors_vec from vec0's shadow tables, or null without a - * readable table. `droppedRows` projects the layout after that many rows are - * deleted without their chunks going away, which is what orphan removal does. + * Chunk layout of the vector table from vec0's shadow tables, and the chunks + * a repack would move: those filled below VEC_REPACK_CHUNK_FILL, except each + * partition's newest. Rows of different partitions never share a chunk, so a + * packed table needs ceil(rows / chunk size) chunks in each partition. With + * `dropOrphans` the plan is for the table as cleanupOrphanedVectors will + * leave it; vec0 drops a chunk with its last row, so a chunk holding only + * orphans is gone from that layout. */ -export function vectorTableLayout(db: Database, droppedRows: number = 0): VectorTableLayout | null { - if (!vecTableReadable(db)) return null; - const chunkRow = db.prepare(`SELECT COUNT(*) AS chunks, MAX(size) AS chunkSize FROM vectors_vec_chunks`).get() as { chunks: number; chunkSize: number | null }; - const stored = (db.prepare(`SELECT COUNT(*) AS c FROM vectors_vec_rowids`).get() as { c: number }).c; - const rows = Math.max(stored - droppedRows, 0); - const chunkSize = chunkRow.chunkSize ?? 0; - const neededChunks = chunkSize > 0 ? Math.ceil(rows / chunkSize) : 0; - const occupancy = chunkRow.chunks > 0 ? neededChunks / chunkRow.chunks : 1; - return { rows, chunks: chunkRow.chunks, neededChunks, occupancy }; -} - -/** Rows of vectors_vec that cleanupOrphanedVectors would delete: orphaned chunks that still hold a vector. */ -function countOrphanedVectorRows(db: Database): number { - if (!vecTableReadable(db)) return 0; - return withLazyContentVectorMigration(db, () => (db.prepare(` - SELECT COUNT(*) AS c - FROM content_vectors cv - WHERE NOT EXISTS (SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1) - AND EXISTS (SELECT 1 FROM vectors_vec_rowids r WHERE r.id = cv.hash || '_' || cv.seq) - `).get() as { c: number }).c); -} - -function liveSlots(chunk: VecChunkRow): number[] { +function vectorRepackPlan(db: Database, dropOrphans: boolean = false): VecRepackPlan | null { + const layout = readableVectorLayout(db); + if (!layout) return null; + const orphansByChunk = new Map(); + if (dropOrphans) { + const rows = db.prepare(` + SELECT r.chunk_id AS chunkId, COUNT(*) AS n + FROM ${VEC_ROWS_TABLE} vr + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.id = vr.collection_id + JOIN ${layout.shadow.rowids} r ON r.rowid = vr.id + WHERE NOT EXISTS (SELECT 1 FROM documents d WHERE d.hash = vr.hash AND d.collection = ci.name AND d.active = 1) + GROUP BY r.chunk_id + `).all() as { chunkId: number; n: number }[]; + for (const row of rows) orphansByChunk.set(row.chunkId, row.n); + } + const chunks = (db.prepare(`SELECT chunk_id AS chunkId, size, partition00 AS partition, validity FROM ${layout.shadow.chunks} ORDER BY chunk_id`).all() as Omit[]) + .map((chunk) => ({ ...chunk, live: liveSlots(chunk).length - (orphansByChunk.get(chunk.chunkId) ?? 0) })) + .filter((chunk) => chunk.live > 0); + const rowsByPartition = new Map(); + const newestByPartition = new Map(); + for (const chunk of chunks) { + rowsByPartition.set(chunk.partition, (rowsByPartition.get(chunk.partition) ?? 0) + chunk.live); + newestByPartition.set(chunk.partition, chunk.chunkId); + } + let rows = 0; + let neededChunks = 0; + for (const chunk of chunks) { + if (newestByPartition.get(chunk.partition) !== chunk.chunkId) continue; + const partitionRows = rowsByPartition.get(chunk.partition)!; + rows += partitionRows; + neededChunks += Math.ceil(partitionRows / chunk.size); + } + const occupancy = chunks.length > 0 ? neededChunks / chunks.length : 1; + const sparse = chunks + .filter((chunk) => newestByPartition.get(chunk.partition) !== chunk.chunkId && chunk.live < chunk.size * VEC_REPACK_CHUNK_FILL) + .map((chunk) => chunk.chunkId); + return { layout: { rows, chunks: chunks.length, neededChunks, occupancy }, sparse }; +} + +/** Whether a plan is worth running: the table's scans read mostly holes and some chunk would move. */ +function repackWanted(plan: VecRepackPlan | null): plan is VecRepackPlan { + return plan !== null && plan.layout.occupancy < VEC_REPACK_BELOW_OCCUPANCY && plan.sparse.length > 0; +} + +/** Chunk layout of the vector table from vec0's shadow tables, or null without a readable table. */ +export function vectorTableLayout(db: Database): VectorTableLayout | null { + return vectorRepackPlan(db)?.layout ?? null; +} + +function liveSlots(chunk: { size: number; validity: Uint8Array }): number[] { const slots: number[] = []; for (let i = 0; i < chunk.size; i++) { if ((chunk.validity[i >> 3]! >> (i & 7)) & 1) slots.push(i); @@ -3333,39 +3366,44 @@ function liveSlots(chunk: VecChunkRow): number[] { } /** - * Compact vectors_vec in place: the live rows of every chunk that is mostly - * holes are deleted and re-inserted, one chunk per IMMEDIATE transaction. - * vec0 puts each insert in the first free slot of its newest chunk and drops - * a chunk once it is empty, so the moved rows fill the tail and the emptied - * chunks vanish. A transaction holds the write lock for at most one chunk of + * Compact the vector table in place: the live rows of every chunk that is + * mostly holes are deleted and re-inserted under their rowid and partition, + * one chunk per IMMEDIATE transaction. vec0 puts each insert in the first + * free slot of its partition's newest chunk and drops a chunk once it is + * empty, so the moved rows fill that partition's tail and the emptied chunks + * vanish; each partition's newest chunk is left alone because that is where + * its rows land. A transaction holds the write lock for at most one chunk of * rows, so concurrent embed and update runs interleave with it, and an * interrupted run leaves a consistent table that the next cleanup finishes. - * A legacy table without a hash_seq key is left for ensureVecTable to rebuild. */ export function repackVectors(db: Database, onChunk?: (moved: number, total: number) => void): VectorTableLayout | null { - if (!vecTableReadable(db)) return null; - const ddl = (db.prepare(`SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'vectors_vec'`).get() as { sql: string }).sql; - if (!ddl.includes("hash_seq")) return vectorTableLayout(db); - const chunks = db.prepare(`SELECT chunk_id AS chunkId, size, validity, rowids FROM vectors_vec_chunks ORDER BY chunk_id`).all() as VecChunkRow[]; - const newest = chunks.length > 0 ? chunks[chunks.length - 1]!.chunkId : -1; - const sparse = chunks.filter((c) => c.chunkId !== newest && liveSlots(c).length < c.size * VEC_REPACK_CHUNK_FILL); - const keyOf = db.prepare(`SELECT id FROM vectors_vec_rowids WHERE rowid = ?`); - const vectorOf = db.prepare(`SELECT embedding FROM vectors_vec WHERE hash_seq = ?`); - const remove = db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`); - const insert = db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`); - sparse.forEach((chunk, i) => { - const rowids = new DataView(chunk.rowids.buffer, chunk.rowids.byteOffset, chunk.rowids.byteLength); + const layout = readableVectorLayout(db); + const plan = vectorRepackPlan(db); + if (!layout || !plan) return null; + const chunkOf = db.prepare(`SELECT chunk_id AS chunkId, size, partition00 AS partition, validity, rowids FROM ${layout.shadow.chunks} WHERE chunk_id = ?`); + const newestOf = db.prepare(`SELECT MAX(chunk_id) AS chunkId FROM ${layout.shadow.chunks} WHERE partition00 = ?`); + const vectorOf = db.prepare(`SELECT embedding FROM ${layout.table} WHERE rowid = ?`); + const remove = db.prepare(`DELETE FROM ${layout.table} WHERE rowid = ?`); + const insert = db.prepare(`INSERT INTO ${layout.table} (rowid, collection_id, embedding) VALUES (?, ?, ?)`); + plan.sparse.forEach((chunkId, i) => { db.transaction(() => { + // Read the chunk again under the write lock: since the plan was made, a + // concurrent embed or update may have emptied it, made it its + // partition's newest, or deleted a row whose rowid now belongs to + // another partition. + const chunk = chunkOf.get(chunkId) as VecChunkRow | undefined; + if (!chunk) return; + if ((newestOf.get(chunk.partition) as { chunkId: number }).chunkId === chunkId) return; + const rowids = new DataView(chunk.rowids.buffer, chunk.rowids.byteOffset, chunk.rowids.byteLength); for (const slot of liveSlots(chunk)) { - const key = (keyOf.get(rowids.getBigInt64(slot * 8, true)) as { id: string } | undefined)?.id; - if (key === undefined) continue; - const row = vectorOf.get(key) as { embedding: Uint8Array } | undefined; + const rowid = rowids.getBigInt64(slot * 8, true); + const row = vectorOf.get(rowid) as { embedding: Uint8Array } | undefined; if (row === undefined) continue; - remove.run(key); - insert.run(key, row.embedding); + remove.run(rowid); + insert.run(rowid, vecInteger(chunk.partition), row.embedding); } }).immediate(); - onChunk?.(i + 1, sparse.length); + onChunk?.(i + 1, plan.sparse.length); }); return vectorTableLayout(db); } @@ -3397,7 +3435,7 @@ export type CleanupStats = { orphanedVectors: number; inactiveDocs: number; orphanedContent: number; - /** Chunk layout of vectors_vec before any repack, null without a vector table. */ + /** Chunk layout of the vector table before any repack, null without a vector table. */ vectorLayout: VectorTableLayout | null; /** True when the vector table was repacked (or, in a preview, would be). */ vectorsRepacked: boolean; @@ -3409,9 +3447,8 @@ export function previewCleanup(db: Database): CleanupStats { const orphanedVectors = countOrphanedVectors(db); const inactiveDocs = (db.prepare(`SELECT COUNT(*) as c FROM documents WHERE active = 0`).get() as { c: number }).c; const orphanedContent = countOrphanedContent(db); - const vectorLayout = vectorTableLayout(db, countOrphanedVectorRows(db)); - const vectorsRepacked = vectorLayout !== null && vectorLayout.occupancy < VEC_REPACK_BELOW_OCCUPANCY; - return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent, vectorLayout, vectorsRepacked }; + const plan = vectorRepackPlan(db, true); + return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent, vectorLayout: plan?.layout ?? null, vectorsRepacked: repackWanted(plan) }; } export type CleanupHooks = { @@ -3427,17 +3464,17 @@ export type CleanupHooks = { export function runCleanup(db: Database, hooks: CleanupHooks = {}): CleanupStats { const cacheCount = deleteLLMCache(db); const orphanedVectors = cleanupOrphanedVectors(db); - const vectorLayout = vectorTableLayout(db); - const vectorsRepacked = vectorLayout !== null && vectorLayout.occupancy < VEC_REPACK_BELOW_OCCUPANCY; - if (vectorsRepacked && vectorLayout) { - hooks.onVectorRepack?.(vectorLayout); + const plan = vectorRepackPlan(db); + const vectorsRepacked = repackWanted(plan); + if (vectorsRepacked) { + hooks.onVectorRepack?.(plan.layout); repackVectors(db); } const inactiveDocs = deleteInactiveDocuments(db); const orphanedContent = cleanupOrphanedContent(db); optimizeDocumentsFts(db); vacuumDatabase(db); - return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent, vectorLayout, vectorsRepacked }; + return { cacheCount, orphanedVectors, inactiveDocs, orphanedContent, vectorLayout: plan?.layout ?? null, vectorsRepacked }; } // ============================================================================= diff --git a/test/cleanup.test.ts b/test/cleanup.test.ts index 0bde3756e..b3d2bc573 100644 --- a/test/cleanup.test.ts +++ b/test/cleanup.test.ts @@ -18,10 +18,12 @@ import { previewCleanup, repackVectors, runCleanup, + searchVec, vectorTableLayout, type Store, } from "../src/store.js"; import type { Database } from "../src/db.js"; +import { VEC_ROWS_TABLE, VEC_TABLE, deletePartitionRows, partitionRowKey, resolveCollectionId, vecInteger } from "../src/vec-layout.js"; let store: Store | null = null; @@ -114,25 +116,26 @@ describe("qmd cleanup reclaim (#550)", () => { describe("qmd cleanup vector repack", () => { const DIMS = 3; + const hashOf = (prefix: string, i: number) => `${prefix}${String(i).padStart(5, "0")}`; + /** - * Inserts `total` vectors, then deletes every one except the indexes in - * `keep` (which also get active documents). vec0 drops a chunk only when it + * Inserts `total` embedded documents into `collection`, then deletes every + * vector row except the indexes in `keep`. vec0 drops a chunk only when it * is completely empty, so keeping one row in each chunk leaves the holes. */ - async function seedVectors(s: Store, total: number, keep: readonly number[]): Promise { + async function seedVectors(s: Store, total: number, keep: readonly number[], collection = "docs", prefix = "vec"): Promise { const now = new Date().toISOString(); s.ensureVecTable(DIMS); s.db.transaction(() => { for (let i = 0; i < total; i++) { - const hash = `vec${String(i).padStart(5, "0")}`; - if (keep.includes(i)) { - insertContent(s.db, hash, `body ${hash}`, now); - insertDocument(s.db, "docs", `${hash}.md`, hash, hash, now, now); - } + const hash = hashOf(prefix, i); + insertContent(s.db, hash, `body ${hash}`, now); + insertDocument(s.db, collection, `${hash}.md`, hash, hash, now, now); s.insertEmbedding(hash, 0, 0, new Float32Array([1, i * 0.001, 0]), "test-model", now, 1); } - const drop = s.db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`); - for (let i = 0; i < total; i++) if (!keep.includes(i)) drop.run(`vec${String(i).padStart(5, "0")}_0`); + const kept = new Set(keep.map((i) => hashOf(prefix, i))); + const rows = s.db.prepare(`SELECT vr.id, vr.hash FROM ${VEC_ROWS_TABLE} vr JOIN vector_collection_ids ci ON ci.id = vr.collection_id WHERE ci.name = ?`).all(collection) as { id: number; hash: string }[]; + deletePartitionRows(s.db, rows.filter((row) => !kept.has(row.hash)).map((row) => row.id)); })(); } @@ -140,8 +143,17 @@ describe("qmd cleanup vector repack", () => { const HOLEY = { total: 1100, keep: [0, 1, 1099] } as const; function nearest(s: Store): string { - const row = s.db.prepare(`SELECT hash_seq FROM vectors_vec WHERE embedding MATCH ? AND k = 1`).get(new Float32Array([1, 0, 0])) as { hash_seq: string }; - return row.hash_seq; + const row = s.db.prepare(`SELECT rowid FROM ${VEC_TABLE} WHERE embedding MATCH ? AND k = 1`).get(new Float32Array([1, 0, 0])) as { rowid: number }; + return partitionRowKey(s.db, row.rowid)!.hash; + } + + /** Hashes whose vector rows are present in the vec0 table, in hash order. */ + function liveHashes(s: Store): string[] { + return (s.db.prepare(`SELECT hash FROM ${VEC_ROWS_TABLE} WHERE id IN (SELECT rowid FROM ${VEC_TABLE}) ORDER BY hash`).all() as { hash: string }[]).map((row) => row.hash); + } + + function partitionCount(s: Store, collection: string): number { + return (s.db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_TABLE} WHERE collection_id = ?`).get(vecInteger(resolveCollectionId(s.db, collection)!)) as { c: number }).c; } test("layout counts the chunks a packed table would need against the chunks in use", async () => { @@ -165,10 +177,8 @@ describe("qmd cleanup vector repack", () => { expect(stats.vectorsRepacked).toBe(true); expect(stats.vectorLayout).toMatchObject({ chunks: 2, neededChunks: 1 }); expect(vectorTableLayout(s.db)).toEqual({ rows: 3, chunks: 1, neededChunks: 1, occupancy: 1 }); - expect(s.db.prepare(`SELECT hash_seq FROM vectors_vec ORDER BY hash_seq`).all()).toEqual([ - { hash_seq: "vec00000_0" }, { hash_seq: "vec00001_0" }, { hash_seq: "vec01099_0" }, - ]); - expect(nearest(s)).toBe("vec00000_0"); + expect(liveHashes(s)).toEqual(["vec00000", "vec00001", "vec01099"]); + expect(nearest(s)).toBe("vec00000"); }); test("runCleanup leaves a packed table alone", async () => { @@ -194,7 +204,7 @@ describe("qmd cleanup vector repack", () => { const failing: Database = { prepare: (sql: string) => { const real = s.db.prepare(sql); - if (!sql.startsWith("INSERT INTO vectors_vec (")) return real; + if (!sql.startsWith(`INSERT INTO ${VEC_TABLE} (`)) return real; return { ...real, get: real.get.bind(real), all: real.all.bind(real), iterate: real.iterate.bind(real), run: () => { throw new Error("injected failure during re-insert"); } }; }, transaction: (fn) => s.db.transaction(fn), @@ -205,10 +215,8 @@ describe("qmd cleanup vector repack", () => { expect(() => repackVectors(failing)).toThrow("injected failure during re-insert"); expect(vectorTableLayout(s.db)).toEqual({ rows: 3, chunks: 2, neededChunks: 1, occupancy: 0.5 }); - expect(nearest(s)).toBe("vec00000_0"); - expect(s.db.prepare(`SELECT hash_seq FROM vectors_vec ORDER BY hash_seq`).all()).toEqual([ - { hash_seq: "vec00000_0" }, { hash_seq: "vec00001_0" }, { hash_seq: "vec01099_0" }, - ]); + expect(nearest(s)).toBe("vec00000"); + expect(liveHashes(s)).toEqual(["vec00000", "vec00001", "vec01099"]); }); test("previewCleanup projects the layout after orphan removal, matching what runCleanup does", async () => { @@ -233,16 +241,72 @@ describe("qmd cleanup vector repack", () => { expect(vectorTableLayout(s.db)).toEqual({ rows: 1024, chunks: 1, neededChunks: 1, occupancy: 1 }); }); - test("a legacy vector table without hash_seq is left alone", async () => { + test("repack moves rows within their own partition and leaves each partition's newest chunk alone", async () => { const s = await openStore(); - s.ensureVecTable(DIMS); - s.db.exec(`DROP TABLE vectors_vec`); - s.db.exec(`CREATE VIRTUAL TABLE vectors_vec USING vec0(hash TEXT PRIMARY KEY, embedding float[${DIMS}] distance_metric=cosine)`); + await seedVectors(s, HOLEY.total, HOLEY.keep, "alpha", "alpha"); + await seedVectors(s, HOLEY.total, HOLEY.keep, "beta", "beta"); + // Rows of different partitions never share a chunk: a packed copy needs one chunk each. + expect(vectorTableLayout(s.db)).toEqual({ rows: 6, chunks: 4, neededChunks: 2, occupancy: 0.5 }); + + const stats = runCleanup(s.db); + + expect(stats.vectorsRepacked).toBe(true); + expect(vectorTableLayout(s.db)).toEqual({ rows: 6, chunks: 2, neededChunks: 2, occupancy: 1 }); + expect(runCleanup(s.db).vectorsRepacked).toBe(false); + expect(partitionCount(s, "alpha")).toBe(3); + expect(partitionCount(s, "beta")).toBe(3); + const alpha = await searchVec(s.db, "ignored", "test-model", 5, "alpha", undefined, [1, 0, 0]); + expect(alpha.map((r) => r.hash).sort()).toEqual(["alpha00000", "alpha00001", "alpha01099"]); + const beta = await searchVec(s.db, "ignored", "test-model", 5, "beta", undefined, [1, 0, 0]); + expect(beta.map((r) => r.hash).sort()).toEqual(["beta00000", "beta00001", "beta01099"]); + }); + + test("a table below the occupancy trigger with no chunk to move is not repacked", async () => { + const s = await openStore(); + // Two chunks at 973 of 1024 slots (above the per-chunk fill) and a newest chunk of 51. + const keep = Array.from({ length: 2099 }, (_, i) => i).filter((i) => !(i <= 50 || (i >= 1024 && i <= 1074))); + await seedVectors(s, 2099, keep); + expect(vectorTableLayout(s.db)).toEqual({ rows: 1997, chunks: 3, neededChunks: 2, occupancy: 2 / 3 }); + + expect(previewCleanup(s.db).vectorsRepacked).toBe(false); + expect(runCleanup(s.db).vectorsRepacked).toBe(false); + }); + + test("previewCleanup leaves out a chunk that orphan removal empties, as runCleanup finds it", async () => { + const s = await openStore(); + await seedVectors(s, 1100, Array.from({ length: 1100 }, (_, i) => i)); s.db.transaction(() => { - for (let i = 0; i < 1100; i++) s.db.prepare(`INSERT INTO vectors_vec (hash, embedding) VALUES (?, ?)`).run(`legacy${i}`, new Float32Array([1, i * 0.001, 0])); - for (let i = 2; i < 1099; i++) s.db.prepare(`DELETE FROM vectors_vec WHERE hash = ?`).run(`legacy${i}`); + for (let i = 0; i < 1024; i++) deactivateDocument(s.db, "docs", `${hashOf("vec", i)}.md`); })(); - expect(repackVectors(s.db)).toEqual({ rows: 3, chunks: 2, neededChunks: 1, occupancy: 0.5 }); + const preview = previewCleanup(s.db); + expect(preview).toMatchObject({ orphanedVectors: 1024, vectorsRepacked: false, vectorLayout: { rows: 76, chunks: 1, neededChunks: 1, occupancy: 1 } }); + + const stats = runCleanup(s.db); + expect(stats.vectorsRepacked).toBe(false); + expect(stats.vectorLayout).toEqual(preview.vectorLayout); + }); + + test("a repack reads each chunk again, so a rowid reused by another partition stays there", async () => { + const s = await openStore(); + // Alpha holds two sparse chunks and a newest one; beta is small and packed. + await seedVectors(s, 2100, [0, 1, 1024, 1025, 2099], "alpha", "alpha"); + await seedVectors(s, 3, [0, 1, 2], "beta", "beta"); + const betaId = resolveCollectionId(s.db, "beta")!; + const reused = (s.db.prepare(`SELECT id FROM ${VEC_ROWS_TABLE} WHERE hash = ?`).get(hashOf("alpha", 1024)) as { id: number }).id; + + repackVectors(s.db, (done) => { + if (done !== 1) return; + // Between two chunk transactions, another writer drops a row of the second + // sparse chunk and a beta embed takes over its rowid. + deletePartitionRows(s.db, [reused]); + s.db.prepare(`INSERT INTO ${VEC_ROWS_TABLE} (id, hash, seq, collection_id) VALUES (?, ?, 0, ?)`).run(reused, "betanew", betaId); + s.db.prepare(`INSERT INTO ${VEC_TABLE} (rowid, collection_id, embedding) VALUES (?, ?, ?)`).run(vecInteger(reused), vecInteger(betaId), new Float32Array([0, 1, 0])); + }); + + const row = s.db.prepare(`SELECT collection_id AS collectionId FROM ${VEC_TABLE} WHERE rowid = ?`).get(vecInteger(reused)) as { collectionId: number }; + expect(row.collectionId).toBe(betaId); + expect(partitionCount(s, "alpha")).toBe(4); + expect(partitionCount(s, "beta")).toBe(4); }); }); From 03bd6f16b4fc732cc2b7670c38d0ca6147e62e91 Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:25:44 -0500 Subject: [PATCH 72/82] test(cleanup): check a partitioned repack keeps vectors, rows and rollback Two partitions of 1,100 documents, of which cleanup removes all but three per partition as orphans. After runCleanup reports a repack, every survivor's stored vector bytes are unchanged, the row in each partition's newest chunk keeps its chunk and offset, the old chunk's rows sit in that newest chunk, and the doctor vector sample check (#978) reports ok. Two more cases: a failed move of a chunk's last row, which vec0 drops inside the transaction, rolls back with its row; and a hash embedded in two collections but active in only one is orphaned in the other alone, so the dry run drops that row and cleanup keeps the active copy. --- test/cleanup-repack-partitions.test.ts | 111 +++++++++++++++++++++++++ test/cleanup.test.ts | 40 +++++++++ 2 files changed, 151 insertions(+) create mode 100644 test/cleanup-repack-partitions.test.ts diff --git a/test/cleanup-repack-partitions.test.ts b/test/cleanup-repack-partitions.test.ts new file mode 100644 index 000000000..ca0cd263f --- /dev/null +++ b/test/cleanup-repack-partitions.test.ts @@ -0,0 +1,111 @@ +/** + * qmd cleanup repacks a partitioned vector table in place: every surviving + * vector keeps its bytes, each partition's newest chunk keeps its rows where + * they are, and the doctor's stored-vector check still passes afterwards. + */ +import { describe, test, expect, afterEach } from "vitest"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createStore, getEmbeddingFingerprint, runCleanup, type Store } from "../src/store.js"; +import { VEC_ROWS_TABLE, VEC_TABLE, resolveCollectionId, storedEmbedding, vecInteger } from "../src/vec-layout.js"; +import { checkEmbeddingVectorSamples } from "../src/cli/qmd.js"; +import { LlamaCpp, setDefaultLlamaCpp } from "../src/llm.js"; + +const MODEL = "repack-model"; +// Two vec0 chunks of 1024 slots per partition: two live rows stay in the old +// chunk and one in the newest, after cleanup removes the rest as orphans. +const TOTAL = 1100; +const KEEP = [0, 1, 1099]; +const COLLECTIONS = ["alpha", "beta"]; + +function vectorFor(i: number): number[] { + return [1, i / TOTAL, 0]; +} + +// Re-embeds a chunk as the vector its document was stored with, so the doctor +// check compares like with like. +class StoredVectorLlm extends LlamaCpp { + async tokenize(text: string) { return new Array(Math.max(1, Math.ceil(text.length / 16))).fill(1); } + async embed(text: string) { + const index = Number(/\d{5}/.exec(text)?.[0]); + return { embedding: vectorFor(index), model: MODEL }; + } +} + +let dir: string | undefined; +let store: Store | undefined; + +afterEach(async () => { + setDefaultLlamaCpp(null); + store?.close(); + store = undefined; + if (dir) await rm(dir, { recursive: true, force: true }); + dir = undefined; +}); + +function hashOf(collection: string, i: number): string { + return `${collection}${String(i).padStart(5, "0")}`; +} + +function seedPartition(s: Store, collection: string): void { + const now = new Date().toISOString(); + const kept = new Set(KEEP); + s.db.transaction(() => { + for (let i = 0; i < TOTAL; i++) { + const hash = hashOf(collection, i); + s.insertContent(hash, `# ${hash}\n\nbody ${hash}`, now); + s.insertDocument(collection, `${hash}.md`, hash, hash, now, now); + s.insertEmbedding(hash, 0, 0, new Float32Array(vectorFor(i)), MODEL, now, 1); + if (!kept.has(i)) s.deactivateDocument(collection, `${hash}.md`); + } + })(); +} + +/** "chunk_id:chunk_offset" of a hash's vector row in the vec0 table. */ +function rowPosition(s: Store, hash: string): string { + const row = s.db.prepare(` + SELECT r.chunk_id AS chunk, r.chunk_offset AS offset + FROM ${VEC_ROWS_TABLE} vr JOIN ${VEC_TABLE}_rowids r ON r.rowid = vr.id + WHERE vr.hash = ? + `).get(hash) as { chunk: number; offset: number }; + return `${row.chunk}:${row.offset}`; +} + +function newestChunk(s: Store, collection: string): number { + const partition = vecInteger(resolveCollectionId(s.db, collection)!); + const row = s.db.prepare(`SELECT MAX(chunk_id) AS chunk FROM ${VEC_TABLE}_chunks WHERE partition00 = ?`) + .get(partition) as { chunk: number }; + return row.chunk; +} + +describe("qmd cleanup repack on a partitioned vector table", () => { + test("keeps every surviving vector's bytes and each partition's newest chunk, and the doctor check still passes", async () => { + dir = await mkdtemp(join(tmpdir(), "qmd-repack-partitions-")); + const s = createStore(join(dir, "index.sqlite")); + store = s; + s.ensureVecTable(3); + for (const collection of COLLECTIONS) seedPartition(s, collection); + const survivors = COLLECTIONS.flatMap(collection => KEEP.map(i => hashOf(collection, i))); + const bytesBefore = new Map(survivors.map(hash => [hash, Array.from(storedEmbedding(s.db, hash, 0)!)])); + const newestBefore = new Map(COLLECTIONS.map(collection => [collection, rowPosition(s, hashOf(collection, 1099))])); + for (const collection of COLLECTIONS) { + expect(rowPosition(s, hashOf(collection, 1099)).split(":")[0]).toBe(String(newestChunk(s, collection))); + } + + const stats = runCleanup(s.db); + + expect(stats.vectorsRepacked).toBe(true); + for (const hash of survivors) { + expect(Array.from(storedEmbedding(s.db, hash, 0)!)).toEqual(bytesBefore.get(hash)); + } + for (const collection of COLLECTIONS) { + expect(rowPosition(s, hashOf(collection, 1099))).toBe(newestBefore.get(collection)); + // The old chunk's two rows moved behind the newest one, which emptied it. + expect(rowPosition(s, hashOf(collection, 0)).split(":")[0]).toBe(String(newestChunk(s, collection))); + } + setDefaultLlamaCpp(new StoredVectorLlm()); + expect(await checkEmbeddingVectorSamples(s.db, MODEL, getEmbeddingFingerprint(MODEL), survivors.length)) + .toMatchObject({ ok: true }); + }); +}); diff --git a/test/cleanup.test.ts b/test/cleanup.test.ts index b3d2bc573..cc1c214b3 100644 --- a/test/cleanup.test.ts +++ b/test/cleanup.test.ts @@ -309,4 +309,44 @@ describe("qmd cleanup vector repack", () => { expect(partitionCount(s, "alpha")).toBe(4); expect(partitionCount(s, "beta")).toBe(4); }); + + test("a failed move of a chunk's last row rolls back the emptied chunk", async () => { + const s = await openStore(); + // The first chunk keeps one row, so moving it empties the chunk inside the transaction. + await seedVectors(s, 1100, [0, 1099]); + const failing: Database = { + prepare: (sql: string) => { + const real = s.db.prepare(sql); + if (!sql.startsWith(`INSERT INTO ${VEC_TABLE} (`)) return real; + return { ...real, get: real.get.bind(real), all: real.all.bind(real), iterate: real.iterate.bind(real), run: () => { throw new Error("injected failure during re-insert"); } }; + }, + transaction: (fn) => s.db.transaction(fn), + exec: (sql: string) => s.db.exec(sql), + loadExtension: (path: string) => s.db.loadExtension(path), + close: () => s.db.close(), + }; + + expect(() => repackVectors(failing)).toThrow("injected failure during re-insert"); + expect(vectorTableLayout(s.db)).toEqual({ rows: 2, chunks: 2, neededChunks: 1, occupancy: 0.5 }); + expect(liveHashes(s)).toEqual(["vec00000", "vec01099"]); + expect(nearest(s)).toBe("vec00000"); + }); + + test("previewCleanup counts a shared hash's row as orphaned only in the collection that dropped it", async () => { + const s = await openStore(); + const now = new Date().toISOString(); + s.ensureVecTable(DIMS); + insertContent(s.db, "sharedhash", "body sharedhash", now); + insertDocument(s.db, "alpha", "shared.md", "shared", "sharedhash", now, now); + insertDocument(s.db, "beta", "shared.md", "shared", "sharedhash", now, now); + s.insertEmbedding("sharedhash", 0, 0, new Float32Array([1, 0, 0]), "test-model", now, 1); + expect(partitionCount(s, "alpha")).toBe(1); + expect(partitionCount(s, "beta")).toBe(1); + deactivateDocument(s.db, "beta", "shared.md"); + + expect(previewCleanup(s.db).vectorLayout).toMatchObject({ rows: 1, chunks: 1 }); + runCleanup(s.db); + expect(partitionCount(s, "alpha")).toBe(1); + expect(partitionCount(s, "beta")).toBe(0); + }); }); From 2262ed5c25b1c1d23cd4a40fa75efc9f02458ed8 Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:47:28 -0500 Subject: [PATCH 73/82] test(update): check a missing collection root keeps its vectors Since #983, `qmd update` deletes the vectors of retired documents as it finishes, so a collection whose directory disappears would lose its embeddings along with its documents. With #990's root check, update warns that the root was not found and leaves the collection's documents active and its vector rows in place. --- test/cli.test.ts | 33 ++++++++++++++++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/test/cli.test.ts b/test/cli.test.ts index de4473b73..4c00979a0 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -6,7 +6,7 @@ */ import { describe, test, expect, beforeAll, afterAll, beforeEach } from "vitest"; -import { chmod, copyFile, mkdtemp, rm, writeFile, mkdir } from "fs/promises"; +import { chmod, copyFile, mkdtemp, rm, writeFile, mkdir, rename } from "fs/promises"; import { existsSync, lstatSync, readFileSync, symlinkSync, writeFileSync, unlinkSync } from "fs"; import { tmpdir } from "os"; import { join, dirname } from "path"; @@ -1424,6 +1424,37 @@ describe("qmd update stale vector rows", () => { } }); + test("update leaves a collection whose directory is missing with its documents and vectors", async () => { + const env = await createIsolatedTestEnv("missing-root"); + const root = join(testDir, `missing-root-${testCounter}`); + await mkdir(root, { recursive: true }); + await writeFile(join(root, "kept.md"), "# Kept\n\nstill here\n"); + const add = await runQmd(["collection", "add", root, "--name", "mounted"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(add.exitCode).toBe(0); + + const store = createStore(env.dbPath); + try { + const { hash } = store.db.prepare(`SELECT hash FROM documents WHERE path = 'kept.md'`).get() as { hash: string }; + store.ensureVecTable(3); + store.insertEmbedding(hash, 0, 0, new Float32Array([1, 2, 3]), "test", new Date().toISOString()); + } finally { + store.close(); + } + await rename(root, `${root}-unmounted`); + + const update = await runQmd(["update"], { dbPath: env.dbPath, configDir: env.configDir }); + expect(update.exitCode).toBe(0); + + const db = openDatabase(env.dbPath); + try { + expect(db.prepare(`SELECT COUNT(*) AS c FROM ${VEC_ROWS_TABLE}`).get()).toEqual({ c: 1 }); + expect(db.prepare(`SELECT active FROM documents WHERE path = 'kept.md'`).get()).toEqual({ active: 1 }); + } finally { + db.close(); + } + expect(update.stderr).toContain("Collection root not found"); + }); + test("a collection that gains an already-embedded document receives its vector rows at the end of update", async () => { const env = await createIsolatedTestEnv("copied-vector-rows"); const firstDir = join(testDir, `copied-vector-rows-first-${testCounter}`); From 09a18b90159630d1d47b4e857998d9b7aede215a Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 11:59:34 -0500 Subject: [PATCH 74/82] docs(changelog): credit #983 and #990 The partitioned vector index bullet had no credit, and #990's bullet credited its issue (#989) rather than the PR. Each now ends with its PR number and author, as qmd's release guidelines ask; #989 stays inline. --- CHANGELOG.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 57c1fa124..8ae49a426 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -37,7 +37,7 @@ so search went empty and a `qmd cleanup` before the drive came back deleted the rows for good. The collection is now reported as not found and its index is left unchanged. Permission and I/O errors still fail with their original - cause. (#989) + cause (#989). #990 (thanks @ParkerRex) - Embedding generation and legacy fingerprint adoption now tokenize documents with the store-selected embedding model instead of the global default. This keeps chunk boundaries aligned with the model that creates and verifies the @@ -112,7 +112,7 @@ admits however many chunks it admits, and a vector search fills its limit even when one long document holds all of the nearest chunks. Downgrading to an older qmd afterwards needs `qmd embed -f`, and a `qmd mcp` server - started before the upgrade must be restarted. + started before the upgrade must be restarted. #983 (thanks @brettdavies) ## [2.8.3] - 2026-08-16 From 454668bb3cc00c3f943b0da23808f229ad067b0a Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 15:11:35 -0500 Subject: [PATCH 75/82] refactor(search): drop the unused per-collection merge helper mergeSearchResultsByScore merged per-collection result lists by score. Keyword search runs one query over the whole collection list, and #983's searchVec merges its per-partition results itself, so nothing calls it at this level. The two comments that pointed at it now state the filepath tie-break directly. --- src/store.ts | 21 +++------------------ 1 file changed, 3 insertions(+), 18 deletions(-) diff --git a/src/store.ts b/src/store.ts index 92f273029..1e4376f3e 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4537,21 +4537,6 @@ function scopedCollectionNames(scope: CollectionScope): string[] | undefined { return names.length > 0 ? names : undefined; } -function mergeSearchResultsByScore(lists: SearchResult[][], limit: number): SearchResult[] { - const best = new Map(); - for (const list of lists) { - for (const r of list) { - const prev = best.get(r.filepath); - if (!prev || r.score > prev.score) best.set(r.filepath, r); - } - } - // Ties go to the smaller filepath, so the order the collections were named - // never decides which of two equal hits survives the limit. - return Array.from(best.values()) - .sort((a, b) => b.score - a.score || compareFilepaths(a, b)) - .slice(0, limit); -} - function compareFilepaths(a: { filepath: string }, b: { filepath: string }): number { return a.filepath < b.filepath ? -1 : a.filepath > b.filepath ? 1 : 0; } @@ -4616,8 +4601,8 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle params.push(...compiledFilter.params); } - // bm25 lower is better; sort ascending, ties by filepath as in - // mergeSearchResultsByScore. + // bm25 lower is better; sort ascending. Ties go to the smaller filepath, so + // the order the collections were named never decides which equal hit survives. sql += ` ORDER BY fm.bm25_score ASC, filepath ASC LIMIT ?`; params.push(limit); @@ -4847,7 +4832,7 @@ export async function searchVec(db: Database, query: string, model: string, limi // Each target yields its own nearest `limit` documents (or all it holds), so // merging them by distance gives the scope's exact nearest `limit`. Ties go - // to the smaller filepath, as in mergeSearchResultsByScore. + // to the smaller filepath, as in searchFTS. return scanTargets .flatMap(target => nearestVecDocuments(scan, resolve, queryVec, limit, target)) .sort((a, b) => a.distance - b.distance || compareFilepaths(a, b)) From e4db0eff2051f2dfeec022288572f577ac927f9f Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 16:08:08 -0500 Subject: [PATCH 76/82] fix(cli): report a file over the size limit as too large, not unreadable Files over 10 MB are skipped with FILE_TOO_LARGE, and the CLI printed them through the unreadable-file branch as "Skipped unreadable file: (FILE_TOO_LARGE)" and counted them as unreadable. They now read "Skipped file over 10 MB: " with their own summary count. --- src/cli/qmd.ts | 7 ++++++- test/cli.test.ts | 4 +++- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 6c5d62d34..0f4f616e5 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -2267,16 +2267,21 @@ function reportMetadataErrors(metadataErrors: number): void { function reportSkippedReads(skippedFiles: { file: string; code: string }[]): void { if (skippedFiles.length === 0) return; + const sizeLimitMb = Math.round(REINDEX_MAX_FILE_SIZE / (1024 * 1024)); for (const skipped of skippedFiles) { if (skipped.code === "OUTSIDE_COLLECTION") { console.warn(`⚠ Skipped file outside collection: ${skipped.file}`); + } else if (skipped.code === "FILE_TOO_LARGE") { + console.warn(`⚠ Skipped file over ${sizeLimitMb} MB: ${skipped.file}`); } else { console.warn(`⚠ Skipped unreadable file: ${skipped.file} (${skipped.code})`); } } const escaped = skippedFiles.filter(f => f.code === "OUTSIDE_COLLECTION").length; - const unreadable = skippedFiles.length - escaped; + const tooLarge = skippedFiles.filter(f => f.code === "FILE_TOO_LARGE").length; + const unreadable = skippedFiles.length - escaped - tooLarge; if (escaped) console.warn(`Skipped ${escaped} file(s) outside the collection root`); + if (tooLarge) console.warn(`Skipped ${tooLarge} file(s) over ${sizeLimitMb} MB`); if (unreadable) console.warn(`Skipped ${unreadable} unreadable file(s)`); } diff --git a/test/cli.test.ts b/test/cli.test.ts index 2b156104d..04aec8e27 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -700,7 +700,9 @@ describe("CLI Add Command", () => { ); expect(exitCode).toBe(0); expect(stdout).toContain("Indexed: 1 new"); - expect(stderr).toContain("big.md (FILE_TOO_LARGE)"); + expect(stderr).toContain("Skipped file over 10 MB: big.md"); + expect(stderr).toContain("Skipped 1 file(s) over 10 MB"); + expect(stderr).not.toContain("unreadable"); }); test("can recreate collection with remove and add", async () => { From f36855cf2c240d98e1461354be74aeb85b088ac2 Mon Sep 17 00:00:00 2001 From: Brett Date: Wed, 30 Sep 2026 16:13:03 -0500 Subject: [PATCH 77/82] perf(store): write a collection scan in short transactions Every write in a scan committed on its own: the content row, the document, its metadata and #962's sync row, one WAL commit each. A first index of a large collection paid that per file, about 176 s for 20,000 files, and a 267,000-file collection's first sync pass wrote 4.7 GB of WAL at about 285 files/s. scanWriteBatch groups the writes of reindexCollection and indexFiles into transactions of up to 500 files or 250 ms, so another writer waits at most one batch. The helpers that open their own transaction nest as savepoints inside it. A failed write rolls back the open batch and rethrows, leaving no transaction behind; the next run indexes the rolled back files again. The same 20,000 files now index in 5.0 s. --- CHANGELOG.md | 4 +++ src/cli/qmd.ts | 6 +++++ src/db.ts | 2 ++ src/store.ts | 63 ++++++++++++++++++++++++++++++++++++++++++++++ test/store.test.ts | 44 ++++++++++++++++++++++++++++++++ 5 files changed, 119 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3727808c3..2d87775e0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -56,6 +56,10 @@ bounds memory on indexes with very large documents. A match past that point is still found, but its snippet, and the passage the reranker scores, come from the start of the document. #962 (thanks @rikvanriel) +- `qmd update` and `qmd collection add` write a collection scan in short + transactions instead of committing every row on its own, so a first index + of 20,000 files takes about 5 s instead of about 3 minutes. #1021 (thanks + @brettdavies) ## [2.8.3] - 2026-08-16 diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 0f4f616e5..eedd2aa27 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -81,6 +81,7 @@ import { createStore, getDefaultDbPath, reindexCollection, + scanWriteBatch, REINDEX_MAX_FILE_SIZE, generateEmbeddings, maybeAdoptLegacyEmbeddingFingerprint, @@ -2132,7 +2133,9 @@ async function indexFiles(pwd?: string, globPattern: string = DEFAULT_GLOB, coll const livePaths = new Set(files.map(f => f.replace(/\\/g, '/'))); const startTime = Date.now(); + const batch = scanWriteBatch(db); for (const relativeFile of files) { + batch.next(); const filepath = getRealPath(resolve(resolvedPwd, relativeFile)); // Store the literal relative path — handelize() is NOT applied at index time. const path = relativeFile.replace(/\\/g, '/'); @@ -2226,11 +2229,14 @@ async function indexFiles(pwd?: string, globPattern: string = DEFAULT_GLOB, coll let removed = 0; for (const path of allActive) { if (!seenPaths.has(path)) { + batch.next(); deactivateDocument(db, collectionName, path); removed++; } } + batch.commit(); + // Clean up orphaned content hashes (content not referenced by any document) const orphanedContent = cleanupOrphanedContent(db); diff --git a/src/db.ts b/src/db.ts index 9273278d9..de23218b0 100644 --- a/src/db.ts +++ b/src/db.ts @@ -127,6 +127,8 @@ export function openDatabase(path: string): Database { * Common subset of the Database interface used throughout QMD. */ export interface Database { + /** Both drivers expose it; true while a transaction or savepoint is open. */ + readonly inTransaction: boolean; exec(sql: string): void; prepare(sql: string): Statement; loadExtension(path: string): void; diff --git a/src/store.ts b/src/store.ts index ae1e4cb96..f618647e8 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1729,6 +1729,43 @@ function retireIndexedFile( syncStateMap.delete(path); } +/** + * Groups the per-file writes of a collection scan into short transactions. + * Committed one by one, a first index of a large collection writes a content + * row, a document, its metadata and a sync row per file, each its own WAL + * commit: 20,000 files took about 176 s that way. A batch closes after + * `maxFiles` files or `maxMs`, so other writers never wait longer than that. + */ +export function scanWriteBatch(db: Database, maxFiles: number = 500, maxMs: number = 250) { + let open = false; + let files = 0; + let startedAt = 0; + const commit = (): void => { + if (!open) return; + open = false; + db.exec("COMMIT"); + }; + return { + /** Call before a file's writes; closes the batch when it is full or old. */ + next(): void { + if (open && (files >= maxFiles || Date.now() - startedAt >= maxMs)) commit(); + if (!open) { + db.exec("BEGIN"); + open = true; + files = 0; + startedAt = Date.now(); + } + files++; + }, + commit, + rollback(): void { + if (!open) return; + open = false; + db.exec("ROLLBACK"); + }, + }; +} + /** * Re-index a single collection by scanning the filesystem and updating the database. * Uses mtime+size fast-path (file_sync_state) to avoid re-reading unchanged files. @@ -1747,6 +1784,28 @@ export async function reindexCollection( ignorePatterns?: string[]; onProgress?: (info: ReindexProgress) => void; } +): Promise { + const batch = scanWriteBatch(store.db); + try { + const result = await reindexCollectionIn(batch, store, collectionPath, globPattern, collectionName, options); + batch.commit(); + return result; + } catch (err) { + batch.rollback(); + throw err; + } +} + +async function reindexCollectionIn( + batch: ReturnType, + store: Store, + collectionPath: string, + globPattern: string, + collectionName: string, + options?: { + ignorePatterns?: string[]; + onProgress?: (info: ReindexProgress) => void; + } ): Promise { const db = store.db; const now = new Date().toISOString(); @@ -1779,6 +1838,7 @@ export async function reindexCollection( const syncStateMap = getFileSyncStateMap(db, collectionName); for (const relativeFile of files) { + batch.next(); const path = normalizePathSeparators(relativeFile); const filepath = getRealPath(resolve(collectionPath, relativeFile)); if (!isPathInsideDir(collectionPath, filepath)) { @@ -1920,12 +1980,15 @@ export async function reindexCollection( let removed = 0; for (const path of allActive) { if (!seenPaths.has(path)) { + batch.next(); deactivateDocument(db, collectionName, path); deleteFileSyncStateForCollection(db, collectionName, path); removed++; } } + batch.commit(); + const orphanedCleaned = cleanupOrphanedContent(db); return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles, metadataErrors }; diff --git a/test/store.test.ts b/test/store.test.ts index a4f26da84..1c529795d 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3146,6 +3146,50 @@ describe("Reindex Collection file sync state (#962)", () => { } }); + test("reindexCollection groups its writes into transactions", async () => { + const store = await createTestStore(); + const collectionPath = join(testDir, `batched-${Date.now()}-${Math.random().toString(36).slice(2)}`); + await mkdir(collectionPath, { recursive: true }); + for (let i = 0; i < 5; i++) await writeFile(join(collectionPath, `d${i}.md`), `# D${i}\n\nbody ${i}\n`); + const seen: boolean[] = []; + + try { + const result = await reindexCollection(store, collectionPath, "**/*.md", "batched", { + onProgress: () => seen.push(store.db.inTransaction), + }); + expect(result.indexed).toBe(5); + expect(seen).toContain(true); + expect(store.db.inTransaction).toBe(false); + const rows = store.db.prepare(`SELECT COUNT(*) AS c FROM file_sync_state WHERE collection = ?`).get("batched") as { c: number }; + expect(rows.c).toBe(5); + } finally { + await rm(collectionPath, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + + test("a failed write rolls back the open batch and leaves no transaction behind", async () => { + const store = await createTestStore(); + const collectionPath = join(testDir, `batch-fail-${Date.now()}-${Math.random().toString(36).slice(2)}`); + await mkdir(collectionPath, { recursive: true }); + for (const name of ["a.md", "b.md", "c.md"]) await writeFile(join(collectionPath, name), `# ${name}\n\nbody\n`); + store.db.exec(`CREATE TRIGGER fail_b BEFORE INSERT ON documents WHEN NEW.path = 'b.md' BEGIN SELECT RAISE(ABORT, 'injected'); END`); + + try { + await expect(reindexCollection(store, collectionPath, "**/*.md", "batch-fail")).rejects.toThrow("injected"); + expect(store.db.inTransaction).toBe(false); + + store.db.exec(`DROP TRIGGER fail_b`); + const retry = await reindexCollection(store, collectionPath, "**/*.md", "batch-fail"); + expect(retry.indexed + retry.unchanged + retry.updated).toBe(3); + const rows = store.db.prepare(`SELECT COUNT(*) AS c FROM file_sync_state WHERE collection = ?`).get("batch-fail") as { c: number }; + expect(rows.c).toBe(3); + } finally { + await rm(collectionPath, { recursive: true, force: true }); + await cleanupTestDb(store); + } + }); + test("files over 10 MB are skipped with FILE_TOO_LARGE and not indexed", async () => { const store = await createTestStore(); const dir = await collectionDir("sync-too-large"); From 6acaa6750bff0d5ea7281ab278f1a5d5bd0f3f73 Mon Sep 17 00:00:00 2001 From: Brett Date: Fri, 2 Oct 2026 10:33:28 -0500 Subject: [PATCH 78/82] test(search): name the filter target `field` in the scoped keyword tests #956 renamed a metadata filter condition's target from `key` to `field`. #953's test of a match ranked below the unfiltered candidate window and the tests of a collection scope combined with a metadata filter still built `{ key: ... }` filters. Such a filter admits no document, so all three searches came back empty. They now name `field`. --- test/metadata-search.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index 30168d86d..a819cbcb6 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -90,7 +90,7 @@ describe("searchFTS with metadata filter", () => { ); const filtered = searchFTS(store.db, "alpha", 5, undefined, { - key: "status", operator: "eq", value: "published", + field: "status", operator: "eq", value: "published", }); expect(filtered.map(r => r.displayPath)).toEqual(["notes/target.md"]); }); @@ -321,7 +321,7 @@ describe("structuredSearch with metadata filter", () => { }); describe("searchFTS with a collection scope and a metadata filter together", () => { - const published: MetadataFilter = { key: "status", operator: "eq", value: "published" }; + const published: MetadataFilter = { field: "status", operator: "eq", value: "published" }; test("returns the in-scope document the filter admits, past both the window and a stronger draft", async () => { // Every noise document outranks both small-collection documents globally, From 6298d7f2a2ea3e11ed1e44685b70ca7fa99a399b Mon Sep 17 00:00:00 2001 From: Brett Date: Fri, 2 Oct 2026 10:38:53 -0500 Subject: [PATCH 79/82] test(search): name the filter target `field` in the partitioned vector tests #956 renamed a metadata filter condition's target from `key` to `field`. #983's searchVec filter tests still built `{ key: ... }` filters, which admit no document, so six of them came back empty and failed. They now name `field`. --- test/metadata-search.test.ts | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/test/metadata-search.test.ts b/test/metadata-search.test.ts index c17731999..a97876e72 100644 --- a/test/metadata-search.test.ts +++ b/test/metadata-search.test.ts @@ -285,7 +285,7 @@ describe("searchVec with metadata filter", () => { expect(filtered.map(r => r.displayPath)).toEqual(["docs/b.md"]); }); - const eligibleOnly: MetadataFilter = { key: "eligible", operator: "eq", value: true }; + const eligibleOnly: MetadataFilter = { field: "eligible", operator: "eq", value: true }; /** An eligible document and an ineligible copy of its content in one collection, one chunk per vector. */ async function insertLongDocumentWithExcludedCopy(vectors: number[][]): Promise { @@ -330,7 +330,7 @@ describe("searchVec with metadata filter", () => { const filtered = await searchVec( store.db, "q", model, 5, "book", undefined, queryEmbedding, undefined, - { key: "eligible", operator: "eq", value: true }, + { field: "eligible", operator: "eq", value: true }, ); expect(filtered).toHaveLength(5); expect(filtered.every(r => r.metadata.eligible === true)).toBe(true); @@ -386,7 +386,7 @@ describe("searchVec with metadata filter", () => { const filtered = await searchVec( store.db, "q", model, 10, "notes", undefined, queryEmbedding, undefined, - { key: "status", operator: "eq", value: "published" }, + { field: "status", operator: "eq", value: "published" }, ); expect(filtered.map(r => r.displayPath)).toEqual(["notes/published-copy.md"]); }); From 06605b2ed90565f59b0c35ef1ac8aa782bafaa1d Mon Sep 17 00:00:00 2001 From: Brett Date: Fri, 2 Oct 2026 10:41:56 -0500 Subject: [PATCH 80/82] perf(search): find filter-eligible collections per document, not per chunk searchVec skips a scoped collection that holds no vector row a metadata filter admits. The query that finds those collections joined each document to all of its vector rows. With a simple filter SQLite applies the filter per document and then walks every row of each eligible document. With a wide filter it scans the vector rows first and applies the filter once per chunk: on #951's near-ceiling test (one document of 20,000 chunks, a filter of 31,490 bindings) the query ran 4.1 s under Bun and 6.9 s under Node per search. The filter now applies once per document, and EXISTS stops at the document's first vector row in the collection, whatever plan the filter leads to. The same query runs in 3 ms under Bun and 5 ms under Node, and returns the same collections. --- src/store.ts | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/src/store.ts b/src/store.ts index 1e4376f3e..c898d4df3 100644 --- a/src/store.ts +++ b/src/store.ts @@ -4724,10 +4724,21 @@ function metadataEligibleRowsSql(filter: MetadataFilter): { sql: string; params: }; } -/** Ids of the collections holding at least one vector row a metadata filter admits. */ +/** + * Ids of the collections holding at least one vector row a metadata filter + * admits. The filter runs once per document and EXISTS probes for a row, so + * the cost follows the documents in scope rather than their chunk count. + */ function metadataEligibleCollections(db: Database, filter: MetadataFilter): Set { - const eligible = metadataEligibleRowsSql(filter); - const rows = db.prepare(`SELECT DISTINCT vr.collection_id AS collectionId ${eligible.sql}`).all(...eligible.params) as { collectionId: number }[]; + const current = compileCurrentMetadataFilter(filter); + const rows = db.prepare(` + SELECT DISTINCT ci.id AS collectionId + FROM documents d + JOIN document_metadata dm ON dm.document_id = d.id + JOIN ${VEC_COLLECTION_IDS_TABLE} ci ON ci.name = d.collection + WHERE d.active = 1 AND ${current.sql} + AND EXISTS (SELECT 1 FROM ${VEC_ROWS_TABLE} vr WHERE vr.hash = d.hash AND vr.collection_id = ci.id) + `).all(...current.params) as { collectionId: number }[]; return new Set(rows.map(row => row.collectionId)); } From 2fe8781d0b6ec5ca53139a1066e31039ee359f61 Mon Sep 17 00:00:00 2001 From: Brett Date: Fri, 2 Oct 2026 11:05:34 -0500 Subject: [PATCH 81/82] fix(store): keep a short body with an embedded NUL whole under the body cap The 256 KiB body cap reads `substr(doc, 1, 262144)`, and SQLite's substr() stops at an embedded NUL. searchFTS, searchVec and getHashesForEmbedding therefore returned only the part of a body before its first NUL, however short the document. Indexing keeps NUL characters, and without the cap these reads return such bodies whole. The three reads now share cappedBodySql: a body whose UTF-8 encoding fits the cap comes back as stored, and only a longer one goes through substr(). A body over 256 KiB with a NUL before the cut still ends at the NUL. Reading 40 bodies of 5 MB takes as long as before. A test per read, each with a short body holding a NUL, fails on the code below, where the body ends at the NUL. Reported in review on #1021. --- src/store.ts | 22 +++++++++++++++++++--- test/store.test.ts | 43 +++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 62 insertions(+), 3 deletions(-) diff --git a/src/store.ts b/src/store.ts index f618647e8..ae7752153 100644 --- a/src/store.ts +++ b/src/store.ts @@ -103,6 +103,22 @@ export function splitGlobMask(mask: string): string[] { } export const DEFAULT_MULTI_GET_MAX_BYTES = 64 * 1024; // 64KB + +/** + * Characters of a document body that search results and getHashesForEmbedding + * return, so one very large document cannot put its whole text on the heap + * per result. + */ +const BODY_CAP_CHARS = 262_144; + +/** + * SQL for the first BODY_CAP_CHARS characters of a body column. substr() + * stops at an embedded NUL, so a body whose UTF-8 encoding fits the cap is + * returned whole and only a longer one goes through substr(). + */ +function cappedBodySql(column: string): string { + return `CASE WHEN length(CAST(${column} AS BLOB)) <= ${BODY_CAP_CHARS} THEN ${column} ELSE substr(${column}, 1, ${BODY_CAP_CHARS}) END`; +} export const DEFAULT_EMBED_MAX_DOCS_PER_BATCH = 64; export const DEFAULT_EMBED_MAX_BATCH_BYTES = 64 * 1024 * 1024; // 64MB export const DEFAULT_EMBED_MAX_DURATION_MS = 30 * 60 * 1000; // 30 minutes; see EmbedOptions.maxDurationMs @@ -4466,7 +4482,7 @@ export function searchFTS(db: Database, query: string, limit: number = 20, colle 'qmd://' || d.collection || '/' || d.path as filepath, d.collection || '/' || d.path as display_path, d.title, - substr(content.doc, 1, 262144) as body, + ${cappedBodySql("content.doc")} as body, d.hash, fm.bm25_score, dm.metadata_json @@ -4679,7 +4695,7 @@ export async function searchVec(db: Database, query: string, model: string, limi 'qmd://' || d.collection || '/' || d.path as filepath, d.collection || '/' || d.path as display_path, d.title, - substr(content.doc, 1, 262144) as body, + ${cappedBodySql("content.doc")} as body, dm.metadata_json FROM content_vectors cv JOIN documents d ON d.hash = cv.hash AND d.active = 1 @@ -4762,7 +4778,7 @@ export function getHashesForEmbedding(db: Database, model: string = DEFAULT_EMBE const fingerprint = getEmbeddingFingerprint(model); return withLazyContentVectorMigration(db, () => { const stmt = db.prepare(` - SELECT d.hash, substr(c.doc, 1, 262144) as body, MIN(d.path) as path + SELECT d.hash, ${cappedBodySql("c.doc")} as body, MIN(d.path) as path FROM documents d JOIN content c ON d.hash = c.hash LEFT JOIN ( diff --git a/test/store.test.ts b/test/store.test.ts index 1c529795d..10e5d5b54 100644 --- a/test/store.test.ts +++ b/test/store.test.ts @@ -3438,6 +3438,49 @@ describe("Reindex Collection file sync state (#962)", () => { await cleanupTestDb(store); } }); + + test("searchFTS returns a short body with an embedded NUL in full", async () => { + const store = await createTestStore(); + try { + const body = "# Nul\n\nzebranul before\u0000after \u00e4\u{1F600}"; + await insertTestDocument(store.db, "docs", { name: "nul", body, displayPath: "nul.md" }); + + const results = store.searchFTS("zebranul", 5); + expect(results).toHaveLength(1); + expect(results[0]!.body).toBe(body); + } finally { + await cleanupTestDb(store); + } + }); + + test("searchVec returns a short body with an embedded NUL in full", async () => { + const store = await createTestStore(); + try { + const body = "# Nul\n\nvector before\u0000after \u00e4\u{1F600}"; + const hash = await hashContent(body); + await insertTestDocument(store.db, "docs", { name: "nul", body, hash, displayPath: "nul.md" }); + store.ensureVecTable(3); + store.insertEmbedding(hash, 0, 0, new Float32Array([1, 0, 0]), "cap-model", new Date().toISOString()); + + const results = await store.searchVec("q", "cap-model", 5, undefined, undefined, [1, 0, 0]); + expect(results).toHaveLength(1); + expect(results[0]!.body).toBe(body); + } finally { + await cleanupTestDb(store); + } + }); + + test("getHashesForEmbedding returns a short body with an embedded NUL in full", async () => { + const store = await createTestStore(); + try { + const body = "# Nul\n\nembed before\u0000after \u00e4\u{1F600}"; + await insertTestDocument(store.db, "docs", { name: "nul", body, displayPath: "nul.md" }); + + expect(store.getHashesForEmbedding().map(row => row.body)).toEqual([body]); + } finally { + await cleanupTestDb(store); + } + }); }); // ============================================================================= From 1d06ffe16484ee23492f67a2a370c9d1656666c0 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Wed, 7 Oct 2026 17:56:24 +0000 Subject: [PATCH 82/82] chore: migrate upstream sync code to strict tsconfig Bring code merged from tobi/qmd (upstream/main 93d211f9) onto the fork's @systemfsoftware/tsconfig base, the same way #2 migrated the rest of src: - noPropertyAccessFromIndexSignature: bracket access for Record reads in parseCliMetadataOptions (src/cli/qmd.ts) and the REST /metadata handler (src/mcp/server.ts). - exactOptionalPropertyTypes: widen ListMetadataOptions fields (src/metadata-store.ts) and reindexCollectionIn options (src/store.ts) to accept explicit undefined. Consumers already treat undefined as absent (`??`, `=== undefined`, truthiness), so no runtime behaviour changes. --- src/cli/qmd.ts | 20 ++++++++++---------- src/mcp/server.ts | 20 ++++++++++---------- src/metadata-store.ts | 18 +++++++++--------- src/store.ts | 4 ++-- 4 files changed, 31 insertions(+), 31 deletions(-) diff --git a/src/cli/qmd.ts b/src/cli/qmd.ts index 853543bef..992de916b 100644 --- a/src/cli/qmd.ts +++ b/src/cli/qmd.ts @@ -2012,27 +2012,27 @@ function collectionMetadata(collectionNames: string[], options: ListMetadataOpti // Parse the discovery-specific flags; exits with usage on a bad value. function parseCliMetadataOptions(values: Record): ListMetadataOptions { const options: ListMetadataOptions = { - match: parseCliMetadataMatch(values.match), - filter: parseCliMetadataFilter(values.filter), + match: parseCliMetadataMatch(values["match"]), + filter: parseCliMetadataFilter(values["filter"]), }; // Discovery windows two dimensions, so the single-window search flags // have no reading here. Point at the flags that do. - if (values.n !== undefined) { + if (values["n"] !== undefined) { console.error("-n is not an option of 'qmd collection metadata'"); console.error("Use --value-limit for values per key, or --key-limit for keys"); process.exit(1); } - if (values.all) { + if (values["all"]) { console.error("--all is not an option of 'qmd collection metadata'"); console.error("Use --all-values, --all-keys, or both"); process.exit(1); } const formatAlias = ["json", "csv", "md", "xml", "files"].find(flag => values[flag]); - const format = typeof values.format === "string" ? values.format.trim().toLowerCase() : undefined; + const format = typeof values["format"] === "string" ? values["format"].trim().toLowerCase() : undefined; if (formatAlias || (format !== undefined && format !== "cli")) { - console.error(`${formatAlias ? `--${formatAlias}` : `--format ${String(values.format)}`} is not supported by 'qmd collection metadata'`); + console.error(`${formatAlias ? `--${formatAlias}` : `--format ${String(values["format"])}`} is not supported by 'qmd collection metadata'`); console.error("This command prints text. Use the SDK, MCP metadata tool, or POST /metadata for structured output"); process.exit(1); } @@ -2059,13 +2059,13 @@ function parseCliMetadataOptions(values: Record): ListMetadataO options.minCount = parsePositiveInteger(values["min-count"], "--min-count"); } - if (values.sort !== undefined) { - if (values.sort !== "count" && values.sort !== "value") { - console.error(`Invalid --sort value: ${String(values.sort)}`); + if (values["sort"] !== undefined) { + if (values["sort"] !== "count" && values["sort"] !== "value") { + console.error(`Invalid --sort value: ${String(values["sort"])}`); console.error("Valid: count, value"); process.exit(1); } - options.sort = values.sort; + options.sort = values["sort"]; } return options; diff --git a/src/mcp/server.ts b/src/mcp/server.ts index ade8828e1..c222170bc 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -1296,13 +1296,13 @@ export async function startMcpHttpServer( // Optional metadata filter — must be an object and a valid filter AST let restFilter: MetadataFilter | undefined; - if (params.filter !== undefined) { - if (typeof params.filter !== "object" || params.filter === null || Array.isArray(params.filter)) { + if (params["filter"] !== undefined) { + if (typeof params["filter"] !== "object" || params["filter"] === null || Array.isArray(params["filter"])) { nodeRes.writeHead(400, { "Content-Type": "application/json" }); nodeRes.end(JSON.stringify({ error: "Invalid field: filter (must be an object)" })); return; } - const filterValidation = validateFilterArgument(params.filter); + const filterValidation = validateFilterArgument(params["filter"]); if (filterValidation.error) { nodeRes.writeHead(400, { "Content-Type": "application/json" }); nodeRes.end(JSON.stringify({ error: filterValidation.error })); @@ -1313,13 +1313,13 @@ export async function startMcpHttpServer( // Optional metadata match, validated the same way against entries let restMatch: MetadataMatch | undefined; - if (params.match !== undefined) { - if (typeof params.match !== "object" || params.match === null || Array.isArray(params.match)) { + if (params["match"] !== undefined) { + if (typeof params["match"] !== "object" || params["match"] === null || Array.isArray(params["match"])) { nodeRes.writeHead(400, { "Content-Type": "application/json" }); nodeRes.end(JSON.stringify({ error: "Invalid field: match (must be an object)" })); return; } - const matchValidation = validateMatchArgument(params.match); + const matchValidation = validateMatchArgument(params["match"]); if (matchValidation.error) { nodeRes.writeHead(400, { "Content-Type": "application/json" }); nodeRes.end(JSON.stringify({ error: matchValidation.error })); @@ -1328,13 +1328,13 @@ export async function startMcpHttpServer( restMatch = matchValidation.match; } - if (params.sort !== undefined && params.sort !== "count" && params.sort !== "value") { + if (params["sort"] !== undefined && params["sort"] !== "count" && params["sort"] !== "value") { nodeRes.writeHead(400, { "Content-Type": "application/json" }); nodeRes.end(JSON.stringify({ error: "Invalid field: sort (must be 'count' or 'value')" })); return; } - if (params.collections !== undefined && !Array.isArray(params.collections)) { + if (params["collections"] !== undefined && !Array.isArray(params["collections"])) { nodeRes.writeHead(400, { "Content-Type": "application/json" }); nodeRes.end(JSON.stringify({ error: "Invalid field: collections (must be an array)" })); return; @@ -1349,7 +1349,7 @@ export async function startMcpHttpServer( } // Use default collections if none specified - const effectiveCollections = params.collections ? params.collections.map(String) : defaultCollectionNames; + const effectiveCollections = params["collections"] ? params["collections"].map(String) : defaultCollectionNames; let result: ListMetadataResult; try { @@ -1357,7 +1357,7 @@ export async function startMcpHttpServer( collection: effectiveCollections.length > 0 ? effectiveCollections : undefined, match: restMatch, filter: restFilter, - sort: params.sort, + sort: params["sort"], ...numberFields.values, }); } catch (err) { diff --git a/src/metadata-store.ts b/src/metadata-store.ts index d97e273b5..2d7257bb2 100644 --- a/src/metadata-store.ts +++ b/src/metadata-store.ts @@ -235,27 +235,27 @@ export function parseMetadataJson(metadataJson: string | null | undefined): Docu export interface ListMetadataOptions { /** Restrict to these collections. Undefined means every collection in the index. */ - collection?: string | string[]; + collection?: string | string[] | undefined; /** * Report only the metadata entries matching this condition. Same grammar as * `filter`, evaluated against each entry: a condition's `field` names the * entry's `key` or `value`. Undefined reports every entry. */ - match?: MetadataMatch; + match?: MetadataMatch | undefined; /** Count only documents matching this filter. Same AST as search. */ - filter?: MetadataFilter; + filter?: MetadataFilter | undefined; /** Keys reported (default 50). `Infinity` removes the window. */ - keyLimit?: number; + keyLimit?: number | undefined; /** Keys skipped before the window, in report order (default 0). */ - keyOffset?: number; + keyOffset?: number | undefined; /** Values reported per key and type (default 10). `Infinity` removes the window. */ - valueLimit?: number; + valueLimit?: number | undefined; /** Values skipped per key and type before the window, in `sort` order (default 0). */ - valueOffset?: number; + valueOffset?: number | undefined; /** Order of values within a key (default "count"). */ - sort?: "count" | "value"; + sort?: "count" | "value" | undefined; /** Drop values held by fewer documents than this (default 1). */ - minCount?: number; + minCount?: number | undefined; } export interface ListMetadataResult { diff --git a/src/store.ts b/src/store.ts index 9dd1c6648..4a46ed863 100644 --- a/src/store.ts +++ b/src/store.ts @@ -1869,8 +1869,8 @@ async function reindexCollectionIn( globPattern: string, collectionName: string, options?: { - ignorePatterns?: string[]; - onProgress?: (info: ReindexProgress) => void; + ignorePatterns?: string[] | undefined; + onProgress?: ((info: ReindexProgress) => void) | undefined; } ): Promise { const db = store.db;