From 2ffec9c71a81522e8e90a2d9e87c0d131fe6e622 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:20:00 -0400 Subject: [PATCH 01/13] fix(search): discard obsolete embedding work and sweep orphan vectors --- .../search/__tests__/file-watcher.test.ts | 456 ++++++++++- .../search/__tests__/hybrid-search.test.ts | 466 ++++++++--- .../search/__tests__/memory-index.test.ts | 191 ++++- .../search/__tests__/memory-recall.test.ts | 23 +- .../search/__tests__/search-index.test.ts | 730 ++++++++++++++++-- src/vault-mcp/search/file-watcher.ts | 234 +++--- src/vault-mcp/search/search-index.ts | 471 +++++++---- 7 files changed, 2100 insertions(+), 471 deletions(-) diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index 6456add24..78df44910 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -1,9 +1,22 @@ import { describe, it, expect, beforeEach, afterEach, vi, onTestFinished } from "vitest" -import { mkdtemp, rm, writeFile, mkdir, rename, unlink, symlink, utimes } from "node:fs/promises" +import { + mkdtemp, + rm, + writeFile, + mkdir, + rename, + unlink, + symlink, + utimes, + readFile, +} from "node:fs/promises" import { join, resolve } from "node:path" import { tmpdir } from "node:os" -import { watch } from "chokidar" +import { watch, FSWatcher } from "chokidar" +import Database from "better-sqlite3" +import * as sqliteVec from "sqlite-vec" import { createSearchIndex } from "../search-index.js" +import { extractPdfText } from "../../obsidian-markdown/pdf.js" import { readdirOrNull, statOrNull } from "../../../utils/fs.js" import type { SearchIndex } from "../search-index.js" import { startFileWatcher } from "../file-watcher.js" @@ -18,6 +31,8 @@ vi.mock("chokidar", { spy: true }) // rescan's synchronous first call, so the delay-default test can observe the // timer firing without waiting on real filesystem I/O. vi.mock("../../../utils/fs.js", { spy: true }) +vi.mock("node:fs/promises", { spy: true }) +vi.mock("../../obsidian-markdown/pdf.js", { spy: true }) let vault: string let index: SearchIndex @@ -190,6 +205,13 @@ describe("file-watcher", REAL_WATCHER_RETRY, () => { it("calls embedNote when indexing a .md file", { timeout: 15000 }, async () => { const embedNoteSpy = vi.spyOn(index, "embedNote") + const capturedVersion = Promise.withResolvers() + const realUpsert = index.upsertNote + vi.spyOn(index, "upsertNote").mockImplementation((params, requestLogger) => { + const sourceVersion = realUpsert(params, requestLogger) + capturedVersion.resolve(sourceVersion) + return sourceVersion + }) await startFileWatcher(vault, index, { stabilityThreshold: 200, pollInterval: 50, @@ -202,10 +224,12 @@ describe("file-watcher", REAL_WATCHER_RETRY, () => { ) await waitFor(() => embedNoteSpy.mock.calls.length > 0) + const sourceVersion = await capturedVersion.promise expect(embedNoteSpy).toHaveBeenCalledWith( { notePath: "embed-test.md", rawContent: "---\ntitle: Embed\n---\n\nEmbed this content\n", + sourceVersion, }, expect.anything(), // logger — runtime child logger, not deterministic ) @@ -325,6 +349,13 @@ describe("file-watcher — file content indexing", REAL_WATCHER_RETRY, () => { fileToolsEnabled: true, }) const embedFileSpy = vi.spyOn(fileIndex, "embedFileContent") + const capturedVersion = Promise.withResolvers() + const realUpsert = fileIndex.upsertFileContent + vi.spyOn(fileIndex, "upsertFileContent").mockImplementation((params, requestLogger) => { + const sourceVersion = realUpsert(params, requestLogger) + capturedVersion.resolve(sourceVersion) + return sourceVersion + }) await startFileWatcher(vault, fileIndex, { stabilityThreshold: 200, @@ -334,13 +365,432 @@ describe("file-watcher — file content indexing", REAL_WATCHER_RETRY, () => { await writeFile(join(vault, "data.csv"), "id,name,value\n1,deploy,active\n", "utf8") await waitFor(() => embedFileSpy.mock.calls.length > 0) + const sourceVersion = await capturedVersion.promise expect(embedFileSpy).toHaveBeenCalledWith( - { filePath: "data.csv" }, + { filePath: "data.csv", sourceVersion }, expect.anything(), // logger — runtime child logger ) }) }) +describe("startFileWatcher — obsolete events and embedding queues", () => { + const createControlledWatcher = async (embedder?: Parameters[1]) => { + const testVault = await mkdtemp(join(tmpdir(), "watcher-events-")) + onTestFinished(() => rm(testVault, { recursive: true })) + const databasePath = join(testVault, "index.db") + const search = createSearchIndex(databasePath, embedder, undefined, { fileToolsEnabled: true }) + const database = new Database(databasePath, { readonly: true }) + sqliteVec.load(database) + onTestFinished(() => { + database.close() + }) + const watcher = new FSWatcher() + const watchMock = vi.mocked(watch).mockReturnValue(watcher) + onTestFinished(async () => { + watchMock.mockRestore() + await watcher.close() + }) + const starting = startFileWatcher(testVault, search) + watcher.emit("ready") + await starting + + const fire = async (event: "change" | "unlink", fileName: string): Promise => { + const handler = watcher.listeners(event)[0] + + if (!handler) throw new Error(`${event} handler was not registered`) + // EventEmitter types listeners as void, but the change handler returns + // its actual promise, so awaiting it observes all indexing work. + await handler(join(testVault, fileName)) + } + return { testVault, search, database, fire } + } + + const delayFirstRead = async (filePath: string) => { + const actualFs = await vi.importActual("node:fs/promises") + const entered = Promise.withResolvers() + const release = Promise.withResolvers() + const delayedPaths = new Set() + vi.mocked(readFile).mockImplementation(async (requestedPath, options) => { + const content = await actualFs.readFile(requestedPath, options) + + if (requestedPath === filePath && !delayedPaths.has(filePath)) { + delayedPaths.add(filePath) + entered.resolve(undefined) + await release.promise + } + return content + }) + onTestFinished(() => vi.mocked(readFile).mockRestore()) + return { entered: entered.promise, release: () => release.resolve(undefined) } + } + + it.each([ + { fileName: "note.md", sourceTable: "notes" }, + { fileName: "content.txt", sourceTable: "file_content" }, + ])("rejects a delayed $fileName read after unlink", async ({ fileName, sourceTable }) => { + const { testVault, database, fire } = await createControlledWatcher() + const filePath = join(testVault, fileName) + await writeFile(filePath, "oldquartz") + await fire("change", fileName) + const delayedRead = await delayFirstRead(filePath) + const lateChange = fire("change", fileName) + await delayedRead.entered + await unlink(filePath) + await fire("unlink", fileName) + + expect(database.prepare(`SELECT path FROM ${sourceTable}`).all()).toEqual([]) + delayedRead.release() + await lateChange + expect(database.prepare(`SELECT path FROM ${sourceTable}`).all()).toEqual([]) + expect(database.prepare("SELECT path FROM notes").all()).toEqual([]) + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + }) + + it.each([ + { fileName: "note.md", sourceTable: "notes" }, + { fileName: "content.txt", sourceTable: "file_content" }, + ])( + "rejects an older $fileName read after the newer handler finishes", + async ({ fileName, sourceTable }) => { + const { testVault, database, fire } = await createControlledWatcher() + const filePath = join(testVault, fileName) + await writeFile(filePath, "oldquartz") + const delayedRead = await delayFirstRead(filePath) + const olderChange = fire("change", fileName) + await delayedRead.entered + await writeFile(filePath, "newopal") + await fire("change", fileName) + + expect(database.prepare(`SELECT path, content FROM ${sourceTable}`).all()).toEqual([ + { path: fileName, content: "newopal" }, + ]) + delayedRead.release() + await olderChange + expect(database.prepare(`SELECT path, content FROM ${sourceTable}`).all()).toEqual([ + { path: fileName, content: "newopal" }, + ]) + }, + ) + + it("keeps the newer event active when an older handler finishes first", async () => { + const { testVault, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "note.md") + await writeFile(filePath, "oldquartz") + const olderRead = await delayFirstRead(filePath) + const olderChange = fire("change", "note.md") + await olderRead.entered + await writeFile(filePath, "newopal") + const newerRead = await delayFirstRead(filePath) + const newerChange = fire("change", "note.md") + await newerRead.entered + olderRead.release() + await olderChange + expect(database.prepare("SELECT path FROM notes").all()).toEqual([]) + newerRead.release() + await newerChange + expect(database.prepare("SELECT path, content FROM notes").all()).toEqual([ + { path: "note.md", content: "newopal" }, + ]) + }) + + it("keeps the committed note when a newer read fails and an older read finishes late", async () => { + const { testVault, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "note.md") + await writeFile(filePath, "committedamber") + await fire("change", "note.md") + await writeFile(filePath, "oldquartz") + const delayedRead = await delayFirstRead(filePath) + const olderChange = fire("change", "note.md") + await delayedRead.entered + const actualFs = await vi.importActual("node:fs/promises") + vi.mocked(readFile).mockImplementation(async (requestedPath, options) => { + if (requestedPath === filePath) throw new Error("controlled read failure") + return actualFs.readFile(requestedPath, options) + }) + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + await fire("change", "note.md") + delayedRead.release() + await olderChange + + expect(errorSpy).toHaveBeenCalledWith("failed to process file change", { + path: "note.md", + error: "[Error]: controlled read failure", + }) + expect(database.prepare("SELECT path, content FROM notes").all()).toEqual([ + { path: "note.md", content: "committedamber" }, + ]) + }) + + it("rejects non-markdown metadata after unlink during stat", async () => { + const { testVault, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "image.png") + await writeFile(filePath, "image data") + await fire("change", "image.png") + const actualFs = + await vi.importActual("../../../utils/fs.js") + const statEntered = Promise.withResolvers() + const releaseStat = Promise.withResolvers() + vi.mocked(statOrNull).mockImplementation(async (requestedPath) => { + const fileStat = await actualFs.statOrNull(requestedPath) + + if (requestedPath === filePath) { + statEntered.resolve(undefined) + await releaseStat.promise + } + return fileStat + }) + onTestFinished(() => vi.mocked(statOrNull).mockRestore()) + const lateChange = fire("change", "image.png") + await statEntered.promise + await unlink(filePath) + await fire("unlink", "image.png") + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + releaseStat.resolve(undefined) + await lateChange + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + }) + + it("rejects a PDF extraction that finishes after unlink", async () => { + const { testVault, database, fire } = await createControlledWatcher() + const { buildMinimalPdf } = await import("../../obsidian-markdown/__tests__/pdf-fixture.js") + const filePath = join(testVault, "doc.pdf") + await writeFile(filePath, buildMinimalPdf()) + const extractionEntered = Promise.withResolvers() + const releaseExtraction = Promise.withResolvers() + const actualPdf = await vi.importActual( + "../../obsidian-markdown/pdf.js", + ) + const extractionSpy = vi.mocked(extractPdfText).mockImplementation(async (pdfData) => { + const extracted = await actualPdf.extractPdfText(pdfData) + extractionEntered.resolve(undefined) + await releaseExtraction.promise + return extracted + }) + onTestFinished(() => extractionSpy.mockRestore()) + const lateChange = fire("change", "doc.pdf") + await extractionEntered.promise + await unlink(filePath) + await fire("unlink", "doc.pdf") + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + releaseExtraction.resolve(undefined) + await lateChange + expect(database.prepare("SELECT path FROM file_content").all()).toEqual([]) + expect(extractionSpy).toHaveBeenCalledTimes(1) + }) + + it.each([ + { + fileName: "note.md", + chunkTable: "note_chunks", + vectorTable: "note_vectors", + pathColumn: "note_path", + prefix: "note", + }, + { + fileName: "content.txt", + chunkTable: "file_content_chunks", + vectorTable: "file_content_vectors", + pathColumn: "file_path", + prefix: "content", + }, + ])( + "skips obsolete queued $fileName jobs and retains the latest queue tail", + async ({ fileName, chunkTable, vectorTable, pathColumn, prefix }) => { + const firstEntered = Promise.withResolvers() + const releaseFirst = Promise.withResolvers() + const replacementEntered = Promise.withResolvers() + const releaseReplacement = Promise.withResolvers() + const finalEntered = Promise.withResolvers() + const releaseFinal = Promise.withResolvers() + const vector = new Float32Array(384).fill(0.1) + const embedder = { + embedText: vi.fn(async (text: string) => { + if (text === `${prefix}\n\noldquartz`) { + firstEntered.resolve(undefined) + return releaseFirst.promise + } + if (text === `${prefix}\n\nnewopal`) { + replacementEntered.resolve(undefined) + return releaseReplacement.promise + } + if (text === `${prefix}\n\nfinalamber`) { + finalEntered.resolve(undefined) + return releaseFinal.promise + } + return vector + }), + embedBatch: vi.fn(async (texts: readonly string[]) => texts.map(() => vector)), + } + const { testVault, database, search, fire } = await createControlledWatcher(embedder) + const filePath = join(testVault, fileName) + const sourceWritten = Promise.withResolvers() + const replacementSourceWritten = Promise.withResolvers() + const realUpsert = fileName.endsWith(".md") ? search.upsertNote : search.upsertFileContent + const upsertName = fileName.endsWith(".md") ? "upsertNote" : "upsertFileContent" + vi.spyOn(search, upsertName).mockImplementation( + ( + params: Parameters[0], + requestLogger: Parameters[1], + ) => { + const sourceVersion = realUpsert(params, requestLogger) + + if (params.rawContent === "queuedberyl") sourceWritten.resolve(undefined) + if (params.rawContent === "newopal") replacementSourceWritten.resolve(undefined) + return sourceVersion + }, + ) + await writeFile(filePath, "oldquartz") + const firstChange = fire("change", fileName) + await firstEntered.promise + await writeFile(filePath, "queuedberyl") + const queuedChange = fire("change", fileName) + await sourceWritten.promise + await unlink(filePath) + await fire("unlink", fileName) + expect( + database + .prepare(`SELECT path FROM ${fileName.endsWith(".md") ? "notes" : "file_content"}`) + .all(), + ).toEqual([]) + await writeFile(filePath, "newopal") + const replacementChange = fire("change", fileName) + await replacementSourceWritten.promise + releaseFirst.resolve(vector) + await firstChange + await queuedChange + await replacementEntered.promise + + expect(embedder.embedText).toHaveBeenCalledTimes(2) + expect(embedder.embedText).toHaveBeenNthCalledWith(1, `${prefix}\n\noldquartz`) + expect(embedder.embedText).toHaveBeenNthCalledWith(2, `${prefix}\n\nnewopal`) + expect(database.prepare(`SELECT chunk_text FROM ${chunkTable}`).all()).toEqual([]) + expect(database.prepare(`SELECT COUNT(*) AS count FROM ${vectorTable}`).get()).toEqual({ + count: 0, + }) + + const finalSourceWritten = Promise.withResolvers() + vi.spyOn(search, upsertName).mockImplementation( + ( + params: Parameters[0], + requestLogger: Parameters[1], + ) => { + const sourceVersion = realUpsert(params, requestLogger) + finalSourceWritten.resolve(undefined) + return sourceVersion + }, + ) + const finalFileEmbedFinished = Promise.withResolvers() + const independentEmbedFinished = Promise.withResolvers() + const independentFileName = fileName.endsWith(".md") ? "sentry.md" : "sentry.txt" + const realFileEmbed = search.embedFileContent + vi.spyOn(search, "embedFileContent").mockImplementation(async (params, requestLogger) => { + await realFileEmbed(params, requestLogger) + if (params.filePath === fileName) finalFileEmbedFinished.resolve(undefined) + if (params.filePath === independentFileName) independentEmbedFinished.resolve(undefined) + }) + await writeFile(filePath, "finalamber") + const finalChange = fire("change", fileName) + await finalSourceWritten.promise + // A different path completes while the replacement stays held, giving + // any incorrectly unqueued final job time to reach its model call. + await writeFile(join(testVault, independentFileName), "independentjade") + await fire("change", independentFileName) + if (!fileName.endsWith(".md")) await independentEmbedFinished.promise + expect(embedder.embedText).toHaveBeenCalledTimes(3) + releaseReplacement.resolve(vector) + await replacementChange + await finalEntered.promise + expect( + database + .prepare(`SELECT chunk_text FROM ${chunkTable} WHERE ${pathColumn} = ?`) + .all(fileName), + ).toEqual([]) + releaseFinal.resolve(vector) + await finalChange + if (!fileName.endsWith(".md")) await finalFileEmbedFinished.promise + const storedChunks = database + .prepare( + `SELECT ${pathColumn} AS path, chunk_text FROM ${chunkTable} WHERE ${pathColumn} = ?`, + ) + .all(fileName) + expect(storedChunks).toEqual([{ path: fileName, chunk_text: `${prefix}\n\nfinalamber` }]) + expect(database.prepare(`SELECT COUNT(*) AS count FROM ${vectorTable}`).get()).toEqual({ + count: 2, + }) + expect(embedder.embedText).toHaveBeenCalledTimes(4) + }, + ) + + it("recovers after a rejected job while another path embeds independently", async () => { + const failingEntered = Promise.withResolvers() + const releaseFailure = Promise.withResolvers() + const recoveredEntered = Promise.withResolvers() + const releaseRecovered = Promise.withResolvers() + const vector = new Float32Array(384).fill(0.1) + const embedder = { + embedText: vi.fn(async (text: string) => { + if (text === "note\n\noldquartz") { + failingEntered.resolve(undefined) + return releaseFailure.promise + } + if (text === "note\n\nnewopal") { + recoveredEntered.resolve(undefined) + return releaseRecovered.promise + } + return vector + }), + embedBatch: vi.fn(async (texts: readonly string[]) => texts.map(() => vector)), + } + const { testVault, database, search, fire } = await createControlledWatcher(embedder) + const sourceWritten = Promise.withResolvers() + const realUpsert = search.upsertNote + vi.spyOn(search, "upsertNote").mockImplementation((params, requestLogger) => { + const sourceVersion = realUpsert(params, requestLogger) + + if (params.rawContent === "newopal") sourceWritten.resolve(undefined) + return sourceVersion + }) + const errorSpy = vi.spyOn(logger, "error") + const debugSpy = vi.spyOn(logger, "debug") + onTestFinished(() => { + errorSpy.mockRestore() + debugSpy.mockRestore() + }) + await writeFile(join(testVault, "note.md"), "oldquartz") + const failingChange = fire("change", "note.md") + await failingEntered.promise + await writeFile(join(testVault, "note.md"), "newopal") + const recoveredChange = fire("change", "note.md") + await sourceWritten.promise + await writeFile(join(testVault, "other.md"), "independentjade") + await fire("change", "other.md") + expect(database.prepare("SELECT note_path, chunk_text FROM note_chunks").all()).toEqual([ + { note_path: "other.md", chunk_text: "other\n\nindependentjade" }, + ]) + releaseFailure.reject(new Error("controlled model failure")) + await failingChange + await recoveredEntered.promise + expect(errorSpy).toHaveBeenCalledWith("failed to process file change", { + path: "note.md", + error: "[Error]: controlled model failure", + }) + expect(debugSpy).toHaveBeenCalledWith("previous embed failed, proceeding with current", { + path: "note.md", + error: "[Error]: controlled model failure", + }) + releaseRecovered.resolve(vector) + await recoveredChange + expect( + database.prepare("SELECT note_path, chunk_text FROM note_chunks ORDER BY note_path").all(), + ).toEqual([ + { note_path: "note.md", chunk_text: "note\n\nnewopal" }, + { note_path: "other.md", chunk_text: "other\n\nindependentjade" }, + ]) + expect(embedder.embedText).toHaveBeenCalledTimes(3) + }) +}) + describe("startFileWatcher — chokidar watch options", () => { type FakeWatcher = { on: (event: string, handler: (...args: unknown[]) => void) => FakeWatcher diff --git a/src/vault-mcp/search/__tests__/hybrid-search.test.ts b/src/vault-mcp/search/__tests__/hybrid-search.test.ts index b7091acfb..d322d21ab 100644 --- a/src/vault-mcp/search/__tests__/hybrid-search.test.ts +++ b/src/vault-mcp/search/__tests__/hybrid-search.test.ts @@ -162,18 +162,24 @@ the Lightsail budget estimates for next quarter. const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteBSourceVersion = hybridIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(1000) }, logger, ) // Embed both notes (same embedding = both match any query equally) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) // Query that matches NOTE_A via FTS ("career goals") and both via vector const { results, search_mode } = await hybridIndex.hybridSearch( @@ -192,17 +198,23 @@ the Lightsail budget estimates for next quarter. const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteBSourceVersion = hybridIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) // Query that matches b.md via FTS ("project ideas CLI") and both via vector const { results } = await hybridIndex.hybridSearch({ query: "project ideas CLI" }, logger) @@ -244,17 +256,23 @@ the Lightsail budget estimates for next quarter. const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteCSourceVersion = hybridIndex.upsertNote( { filePath: "c.md", rawContent: NOTE_C, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "c.md", rawContent: NOTE_C }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteCSourceVersion, notePath: "c.md", rawContent: NOTE_C }, + logger, + ) // Query that doesn't match any note via FTS — results are vector-only const { results } = await hybridIndex.hybridSearch({ query: "zzz_no_fts_match" }, logger) @@ -289,7 +307,7 @@ tags: [test] Content about deployment costs and infrastructure. ` - hybridIndex.upsertNote( + const insideNoteSourceVersion = hybridIndex.upsertNote( { filePath: "Work/inside.md", rawContent: noteInFolder, @@ -297,7 +315,7 @@ Content about deployment costs and infrastructure. }, logger, ) - hybridIndex.upsertNote( + const outsideNoteSourceVersion = hybridIndex.upsertNote( { filePath: "Personal/outside.md", rawContent: noteOutsideFolder, @@ -306,9 +324,20 @@ Content about deployment costs and infrastructure. logger, ) - await hybridIndex.embedNote({ notePath: "Work/inside.md", rawContent: noteInFolder }, logger) await hybridIndex.embedNote( - { notePath: "Personal/outside.md", rawContent: noteOutsideFolder }, + { + sourceVersion: insideNoteSourceVersion, + notePath: "Work/inside.md", + rawContent: noteInFolder, + }, + logger, + ) + await hybridIndex.embedNote( + { + sourceVersion: outsideNoteSourceVersion, + notePath: "Personal/outside.md", + rawContent: noteOutsideFolder, + }, logger, ) @@ -332,7 +361,7 @@ title: Inside Folder --- Content about deployment costs and infrastructure. ` - hybridIndex.upsertNote( + const insideNoteSourceVersion = hybridIndex.upsertNote( { filePath: "Work/inside.md", rawContent: noteInFolder, @@ -340,10 +369,17 @@ Content about deployment costs and infrastructure. }, logger, ) - await hybridIndex.embedNote({ notePath: "Work/inside.md", rawContent: noteInFolder }, logger) + await hybridIndex.embedNote( + { + sourceVersion: insideNoteSourceVersion, + notePath: "Work/inside.md", + rawContent: noteInFolder, + }, + logger, + ) // An equally close note outside the folder proves the filter is // applied at all, not merely that the inside note survives - hybridIndex.upsertNote( + const outsideNoteSourceVersion = hybridIndex.upsertNote( { filePath: "Personal/outside.md", rawContent: noteInFolder, @@ -352,7 +388,11 @@ Content about deployment costs and infrastructure. logger, ) await hybridIndex.embedNote( - { notePath: "Personal/outside.md", rawContent: noteInFolder }, + { + sourceVersion: outsideNoteSourceVersion, + notePath: "Personal/outside.md", + rawContent: noteInFolder, + }, logger, ) @@ -371,17 +411,23 @@ Content about deployment costs and infrastructure. const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteCSourceVersion = hybridIndex.upsertNote( { filePath: "c.md", rawContent: NOTE_C, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "c.md", rawContent: NOTE_C }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteCSourceVersion, notePath: "c.md", rawContent: NOTE_C }, + logger, + ) const { results } = await hybridIndex.hybridSearch( { query: "deployment infrastructure", filters: { tags: ["work"] } }, @@ -398,17 +444,23 @@ Content about deployment costs and infrastructure. const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteCSourceVersion = hybridIndex.upsertNote( { filePath: "c.md", rawContent: NOTE_C, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "c.md", rawContent: NOTE_C }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteCSourceVersion, notePath: "c.md", rawContent: NOTE_C }, + logger, + ) const { results } = await hybridIndex.hybridSearch( { query: "deployment timeline", filters: { type: "meeting" } }, @@ -432,7 +484,7 @@ created: ${createdDate} Content about quarterly planning and roadmaps. ` - hybridIndex.upsertNote( + const onDayNoteSourceVersion = hybridIndex.upsertNote( { filePath: "on-day.md", rawContent: noteCreatedOn("2026-03-10"), @@ -440,7 +492,7 @@ Content about quarterly planning and roadmaps. }, logger, ) - hybridIndex.upsertNote( + const otherDayNoteSourceVersion = hybridIndex.upsertNote( { filePath: "other-day.md", rawContent: noteCreatedOn("2026-03-11"), @@ -450,11 +502,19 @@ Content about quarterly planning and roadmaps. ) await hybridIndex.embedNote( - { notePath: "on-day.md", rawContent: noteCreatedOn("2026-03-10") }, + { + sourceVersion: onDayNoteSourceVersion, + notePath: "on-day.md", + rawContent: noteCreatedOn("2026-03-10"), + }, logger, ) await hybridIndex.embedNote( - { notePath: "other-day.md", rawContent: noteCreatedOn("2026-03-11") }, + { + sourceVersion: otherDayNoteSourceVersion, + notePath: "other-day.md", + rawContent: noteCreatedOn("2026-03-11"), + }, logger, ) @@ -481,7 +541,7 @@ title: Timestamped Content about quarterly planning and roadmaps. ` - hybridIndex.upsertNote( + const duringNoteSourceVersion = hybridIndex.upsertNote( { filePath: "during.md", rawContent: noteBody, @@ -489,7 +549,7 @@ Content about quarterly planning and roadmaps. }, logger, ) - hybridIndex.upsertNote( + const dayAfterNoteSourceVersion = hybridIndex.upsertNote( { filePath: "day-after.md", rawContent: noteBody, @@ -498,8 +558,18 @@ Content about quarterly planning and roadmaps. logger, ) - await hybridIndex.embedNote({ notePath: "during.md", rawContent: noteBody }, logger) - await hybridIndex.embedNote({ notePath: "day-after.md", rawContent: noteBody }, logger) + await hybridIndex.embedNote( + { sourceVersion: duringNoteSourceVersion, notePath: "during.md", rawContent: noteBody }, + logger, + ) + await hybridIndex.embedNote( + { + sourceVersion: dayAfterNoteSourceVersion, + notePath: "day-after.md", + rawContent: noteBody, + }, + logger, + ) const { results } = await hybridIndex.hybridSearch( { @@ -532,22 +602,31 @@ Content about quarterly planning and roadmaps. const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteBSourceVersion = hybridIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteCSourceVersion = hybridIndex.upsertNote( { filePath: "c.md", rawContent: NOTE_C, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) - await hybridIndex.embedNote({ notePath: "c.md", rawContent: NOTE_C }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteCSourceVersion, notePath: "c.md", rawContent: NOTE_C }, + logger, + ) const { results } = await hybridIndex.hybridSearch({ query: "project", limit: 1 }, logger) @@ -580,11 +659,14 @@ This section is about monitoring and observability patterns. We should track latency and error rates across all services. ` - hybridIndex.upsertNote( + const longNoteSourceVersion = hybridIndex.upsertNote( { filePath: "long.md", rawContent: longNote, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "long.md", rawContent: longNote }, logger) + await hybridIndex.embedNote( + { sourceVersion: longNoteSourceVersion, notePath: "long.md", rawContent: longNote }, + logger, + ) const { results } = await hybridIndex.hybridSearch({ query: "deployment" }, logger) @@ -609,7 +691,7 @@ tags: [reference] The main content discusses RESTful API design and GraphQL alternatives. ` - hybridIndex.upsertNote( + const refNoteSourceVersion = hybridIndex.upsertNote( { filePath: "ref.md", rawContent: noteWithCallout, @@ -617,7 +699,10 @@ The main content discusses RESTful API design and GraphQL alternatives. }, logger, ) - await hybridIndex.embedNote({ notePath: "ref.md", rawContent: noteWithCallout }, logger) + await hybridIndex.embedNote( + { sourceVersion: refNoteSourceVersion, notePath: "ref.md", rawContent: noteWithCallout }, + logger, + ) const { results } = await hybridIndex.hybridSearch( { query: "API design patterns", include_leading_callout: true }, @@ -640,17 +725,23 @@ The main content discusses RESTful API design and GraphQL alternatives. const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const noteASourceVersion = hybridIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - hybridIndex.upsertNote( + const noteCSourceVersion = hybridIndex.upsertNote( { filePath: "c.md", rawContent: NOTE_C, fileStat: testStat(1000) }, logger, ) - await hybridIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await hybridIndex.embedNote({ notePath: "c.md", rawContent: NOTE_C }, logger) + await hybridIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await hybridIndex.embedNote( + { sourceVersion: noteCSourceVersion, notePath: "c.md", rawContent: NOTE_C }, + logger, + ) const { results } = await hybridIndex.hybridSearch( { @@ -687,7 +778,7 @@ status: archived This project is no longer maintained but had deployment infrastructure. ` - hybridIndex.upsertNote( + const activeNoteSourceVersion = hybridIndex.upsertNote( { filePath: "active.md", rawContent: noteWithProperty, @@ -695,7 +786,7 @@ This project is no longer maintained but had deployment infrastructure. }, logger, ) - hybridIndex.upsertNote( + const archivedNoteSourceVersion = hybridIndex.upsertNote( { filePath: "archived.md", rawContent: noteWithoutProperty, @@ -704,9 +795,20 @@ This project is no longer maintained but had deployment infrastructure. logger, ) - await hybridIndex.embedNote({ notePath: "active.md", rawContent: noteWithProperty }, logger) await hybridIndex.embedNote( - { notePath: "archived.md", rawContent: noteWithoutProperty }, + { + sourceVersion: activeNoteSourceVersion, + notePath: "active.md", + rawContent: noteWithProperty, + }, + logger, + ) + await hybridIndex.embedNote( + { + sourceVersion: archivedNoteSourceVersion, + notePath: "archived.md", + rawContent: noteWithoutProperty, + }, logger, ) @@ -738,7 +840,7 @@ tags: [test] This is a note with many words that should be truncated when using a small snippet token limit for vector-only results. ` - hybridIndex.upsertNote( + const verboseNoteSourceVersion = hybridIndex.upsertNote( { filePath: "verbose.md", rawContent: verboseNote, @@ -746,7 +848,14 @@ This is a note with many words that should be truncated when using a small snipp }, logger, ) - await hybridIndex.embedNote({ notePath: "verbose.md", rawContent: verboseNote }, logger) + await hybridIndex.embedNote( + { + sourceVersion: verboseNoteSourceVersion, + notePath: "verbose.md", + rawContent: verboseNote, + }, + logger, + ) // Query that won't match via FTS — forces vector-only result path const { results } = await hybridIndex.hybridSearch( @@ -772,16 +881,22 @@ This is a note with many words that should be truncated when using a small snipp const mockEmbedder = createHybridMockEmbedder() const mockReranker = createMockReranker([0.9, 0.1]) const rerankedIndex = createSearchIndex(":memory:", mockEmbedder, mockReranker) - rerankedIndex.upsertNote( + const noteASourceVersion = rerankedIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - rerankedIndex.upsertNote( + const noteBSourceVersion = rerankedIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(2000) }, logger, ) - await rerankedIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await rerankedIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await rerankedIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await rerankedIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) const { reranked } = await rerankedIndex.hybridSearch({ query: "career goals" }, logger) expect(reranked).toBe(true) @@ -802,11 +917,14 @@ This is a note with many words that should be truncated when using a small snipp it("sets reranked to false when embedder exists but no reranker", async () => { const mockEmbedder = createHybridMockEmbedder() const noRerankerIndex = createSearchIndex(":memory:", mockEmbedder) - noRerankerIndex.upsertNote( + const noteASourceVersion = noRerankerIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - await noRerankerIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) + await noRerankerIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) const { reranked } = await noRerankerIndex.hybridSearch({ query: "career goals" }, logger) expect(reranked).toBe(false) @@ -818,16 +936,22 @@ This is a note with many words that should be truncated when using a small snipp rerankPairs: vi.fn().mockRejectedValue(new Error("model failed to load")), } const failIndex = createSearchIndex(":memory:", mockEmbedder, failingReranker) - failIndex.upsertNote( + const noteASourceVersion = failIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - failIndex.upsertNote( + const noteBSourceVersion = failIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(2000) }, logger, ) - await failIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await failIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await failIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await failIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) const warnSpy = vi.spyOn(logger, "warn") const { results, reranked } = await failIndex.hybridSearch({ query: "career goals" }, logger) @@ -845,16 +969,22 @@ This is a note with many words that should be truncated when using a small snipp const mockEmbedder = createHybridMockEmbedder() const mockReranker = createMockReranker([0.9, 0.1]) const rerankIndex = createSearchIndex(":memory:", mockEmbedder, mockReranker) - rerankIndex.upsertNote( + const noteASourceVersion = rerankIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - rerankIndex.upsertNote( + const noteBSourceVersion = rerankIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(2000) }, logger, ) - await rerankIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await rerankIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await rerankIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await rerankIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) await rerankIndex.hybridSearch({ query: "career goals" }, logger) @@ -873,29 +1003,41 @@ This is a note with many words that should be truncated when using a small snipp // Reranker strongly favors b.md (index 1) over a.md (index 0) const mockReranker = createMockReranker([0.1, 0.9]) const rerankIndex = createSearchIndex(":memory:", mockEmbedder, mockReranker) - rerankIndex.upsertNote( + const noteASourceVersion = rerankIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - rerankIndex.upsertNote( + const noteBSourceVersion = rerankIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(2000) }, logger, ) - await rerankIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await rerankIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await rerankIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await rerankIndex.embedNote( + { sourceVersion: noteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) // Get RRF-only scores (no reranker) const rrfOnlyIndex = createSearchIndex(":memory:", mockEmbedder) - rrfOnlyIndex.upsertNote( + const rrfOnlyNoteASourceVersion = rrfOnlyIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - rrfOnlyIndex.upsertNote( + const rrfOnlyNoteBSourceVersion = rrfOnlyIndex.upsertNote( { filePath: "b.md", rawContent: NOTE_B, fileStat: testStat(2000) }, logger, ) - await rrfOnlyIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) - await rrfOnlyIndex.embedNote({ notePath: "b.md", rawContent: NOTE_B }, logger) + await rrfOnlyIndex.embedNote( + { sourceVersion: rrfOnlyNoteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) + await rrfOnlyIndex.embedNote( + { sourceVersion: rrfOnlyNoteBSourceVersion, notePath: "b.md", rawContent: NOTE_B }, + logger, + ) const { results: rrfResults } = await rrfOnlyIndex.hybridSearch( { query: "career goals" }, @@ -921,11 +1063,14 @@ This is a note with many words that should be truncated when using a small snipp const singleResultIndex = createSearchIndex(":memory:", mockEmbedder, mockReranker) // Only index one note so only one result can appear - singleResultIndex.upsertNote( + const noteASourceVersion = singleResultIndex.upsertNote( { filePath: "a.md", rawContent: NOTE_A, fileStat: testStat(1000) }, logger, ) - await singleResultIndex.embedNote({ notePath: "a.md", rawContent: NOTE_A }, logger) + await singleResultIndex.embedNote( + { sourceVersion: noteASourceVersion, notePath: "a.md", rawContent: NOTE_A }, + logger, + ) const { reranked, results } = await singleResultIndex.hybridSearch( { query: "career goals" }, @@ -961,7 +1106,7 @@ describe("hybridSearch — file content vector search", () => { }) // Seed a note (FTS + vector) and a text file (FTS + vector) - fileIndex.upsertNote( + const careerNoteSourceVersion = fileIndex.upsertNote( { filePath: "notes/career.md", rawContent: "---\ntitle: Career\ntags: [personal]\n---\n\nCareer goals and aspirations.\n", @@ -971,6 +1116,7 @@ describe("hybridSearch — file content vector search", () => { ) await fileIndex.embedNote( { + sourceVersion: careerNoteSourceVersion, notePath: "notes/career.md", rawContent: "---\ntitle: Career\ntags: [personal]\n---\n\nCareer goals and aspirations.\n", }, @@ -978,7 +1124,7 @@ describe("hybridSearch — file content vector search", () => { ) fileIndex.upsertNonMdFile("docs/guide.txt", 200) - fileIndex.upsertFileContent( + const guideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "docs/guide.txt", rawContent: "Comprehensive deployment guide covering infrastructure and monitoring setup.", @@ -986,7 +1132,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "docs/guide.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: guideFileSourceVersion, filePath: "docs/guide.txt" }, + logger, + ) // Query that matches the text file via FTS ("deployment guide") and both // items via vector (all embeddings are identical) @@ -1010,7 +1159,7 @@ describe("hybridSearch — file content vector search", () => { // Seed ONLY a text file with no lexical overlap with the query fileIndex.upsertNonMdFile("specs/api-spec.yaml", 300) - fileIndex.upsertFileContent( + const apiSpecFileSourceVersion = fileIndex.upsertFileContent( { filePath: "specs/api-spec.yaml", rawContent: @@ -1019,7 +1168,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "specs/api-spec.yaml" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: apiSpecFileSourceVersion, filePath: "specs/api-spec.yaml" }, + logger, + ) // Query shares no stems with the file content — FTS returns nothing, // but the mock embedder returns identical embeddings so KNN matches @@ -1043,7 +1195,7 @@ describe("hybridSearch — file content vector search", () => { fileToolsEnabled: true, }) - fileIndex.upsertNote( + const careerNoteSourceVersion = fileIndex.upsertNote( { filePath: "notes/career.md", rawContent: "---\ntitle: Career\n---\n\nCareer goals and aspirations.\n", @@ -1053,13 +1205,14 @@ describe("hybridSearch — file content vector search", () => { ) await fileIndex.embedNote( { + sourceVersion: careerNoteSourceVersion, notePath: "notes/career.md", rawContent: "---\ntitle: Career\n---\n\nCareer goals and aspirations.\n", }, logger, ) fileIndex.upsertNonMdFile("docs/guide.txt", 200) - fileIndex.upsertFileContent( + const guideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "docs/guide.txt", rawContent: "Deployment guide covering infrastructure setup.", @@ -1067,7 +1220,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "docs/guide.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: guideFileSourceVersion, filePath: "docs/guide.txt" }, + logger, + ) const warnSpy = vi.spyOn(logger, "warn") onTestFinished(() => warnSpy.mockRestore()) @@ -1089,7 +1245,7 @@ describe("hybridSearch — file content vector search", () => { const mockEmbedder = createHybridMockEmbedder() const hybridIndex = createSearchIndex(":memory:", mockEmbedder) - hybridIndex.upsertNote( + const careerNoteSourceVersion = hybridIndex.upsertNote( { filePath: "notes/career.md", rawContent: "---\ntitle: Career\n---\n\nCareer goals and aspirations.\n", @@ -1099,6 +1255,7 @@ describe("hybridSearch — file content vector search", () => { ) await hybridIndex.embedNote( { + sourceVersion: careerNoteSourceVersion, notePath: "notes/career.md", rawContent: "---\ntitle: Career\n---\n\nCareer goals and aspirations.\n", }, @@ -1131,7 +1288,7 @@ describe("hybridSearch — file content vector search", () => { }) fileIndex.upsertNonMdFile("Docs/inside.txt", 100) - fileIndex.upsertFileContent( + const insideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "Docs/inside.txt", rawContent: "Weekly operations checklist for the deployment crew.", @@ -1139,12 +1296,15 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "Docs/inside.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: insideFileSourceVersion, filePath: "Docs/inside.txt" }, + logger, + ) // "Docs2" starts with "Docs" as a bare string — only a segment-boundary // check (folder + "/") keeps it out of a "Docs" filter fileIndex.upsertNonMdFile("Docs2/outside.txt", 100) - fileIndex.upsertFileContent( + const outsideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "Docs2/outside.txt", rawContent: "Weekly operations checklist for the deployment crew.", @@ -1152,7 +1312,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "Docs2/outside.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: outsideFileSourceVersion, filePath: "Docs2/outside.txt" }, + logger, + ) // No lexical overlap with the seeded content — both files can only // surface through the vector leg (identical mock embeddings), so the @@ -1172,7 +1335,7 @@ describe("hybridSearch — file content vector search", () => { }) fileIndex.upsertNonMdFile("Docs/inside.txt", 100) - fileIndex.upsertFileContent( + const insideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "Docs/inside.txt", rawContent: "Weekly operations checklist for the deployment crew.", @@ -1180,12 +1343,15 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "Docs/inside.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: insideFileSourceVersion, filePath: "Docs/inside.txt" }, + logger, + ) // An equally close file outside the folder proves the filter is applied // at all — without it, "folder ignored" and "folder matched" look alike fileIndex.upsertNonMdFile("Archive/outside.txt", 100) - fileIndex.upsertFileContent( + const outsideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "Archive/outside.txt", rawContent: "Weekly operations checklist for the deployment crew.", @@ -1193,7 +1359,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "Archive/outside.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: outsideFileSourceVersion, filePath: "Archive/outside.txt" }, + logger, + ) // Vector-only hit (no lexical overlap) + a folder filter that differs // only by case — must pass, as it would on the SQL LIKE leg @@ -1211,7 +1380,7 @@ describe("hybridSearch — file content vector search", () => { fileToolsEnabled: true, }) - fileIndex.upsertNote( + const taggedNoteSourceVersion = fileIndex.upsertNote( { filePath: "notes/tagged.md", rawContent: "---\ntitle: Tagged\ntags: [important]\n---\n\nTagged content here.\n", @@ -1221,6 +1390,7 @@ describe("hybridSearch — file content vector search", () => { ) await fileIndex.embedNote( { + sourceVersion: taggedNoteSourceVersion, notePath: "notes/tagged.md", rawContent: "---\ntitle: Tagged\ntags: [important]\n---\n\nTagged content here.\n", }, @@ -1228,7 +1398,7 @@ describe("hybridSearch — file content vector search", () => { ) fileIndex.upsertNonMdFile("data/report.csv", 100) - fileIndex.upsertFileContent( + const reportFileSourceVersion = fileIndex.upsertFileContent( { filePath: "data/report.csv", rawContent: "tagged content in a file", @@ -1236,7 +1406,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "data/report.csv" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: reportFileSourceVersion, filePath: "data/report.csv" }, + logger, + ) // Tag filter is note-specific — file content (FTS and vector) should be excluded const { results } = await fileIndex.hybridSearch( @@ -1255,7 +1428,7 @@ describe("hybridSearch — file content vector search", () => { fileToolsEnabled: true, }) - fileIndex.upsertNote( + const taggedNoteSourceVersion = fileIndex.upsertNote( { filePath: "notes/tagged.md", rawContent: "---\ntitle: Tagged\ntags: [important]\n---\n\nTagged content here.\n", @@ -1265,6 +1438,7 @@ describe("hybridSearch — file content vector search", () => { ) await fileIndex.embedNote( { + sourceVersion: taggedNoteSourceVersion, notePath: "notes/tagged.md", rawContent: "---\ntitle: Tagged\ntags: [important]\n---\n\nTagged content here.\n", }, @@ -1272,7 +1446,7 @@ describe("hybridSearch — file content vector search", () => { ) fileIndex.upsertNonMdFile("data/report.csv", 100) - fileIndex.upsertFileContent( + const reportFileSourceVersion = fileIndex.upsertFileContent( { filePath: "data/report.csv", rawContent: "tagged content in a file", @@ -1280,7 +1454,10 @@ describe("hybridSearch — file content vector search", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "data/report.csv" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: reportFileSourceVersion, filePath: "data/report.csv" }, + logger, + ) // Empty collections constrain nothing in the note leg, so the file // must stay in the merged results exactly as with no filters at all. @@ -1410,11 +1587,14 @@ describe("hybridSearch — folder-scoped vector candidate window", () => { for (const noteNumber of [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]) { const notePath = `Archive/outside-${noteNumber}.md` const rawContent = `# Outside ${noteNumber}\n\nUnrelated archive material.\n` - hybridIndex.upsertNote({ filePath: notePath, rawContent, fileStat: testStat(1000) }, logger) - await hybridIndex.embedNote({ notePath, rawContent }, logger) + const sourceVersion = hybridIndex.upsertNote( + { filePath: notePath, rawContent, fileStat: testStat(1000) }, + logger, + ) + await hybridIndex.embedNote({ sourceVersion, notePath, rawContent }, logger) } const insideContent = "# Inside\n\nThe one note that lives inside Work.\n" - hybridIndex.upsertNote( + const insideNoteSourceVersion = hybridIndex.upsertNote( { filePath: "Work/inside.md", rawContent: insideContent, @@ -1422,7 +1602,14 @@ describe("hybridSearch — folder-scoped vector candidate window", () => { }, logger, ) - await hybridIndex.embedNote({ notePath: "Work/inside.md", rawContent: insideContent }, logger) + await hybridIndex.embedNote( + { + sourceVersion: insideNoteSourceVersion, + notePath: "Work/inside.md", + rawContent: insideContent, + }, + logger, + ) // No lexical overlap with anything seeded — vector legs only const { results, search_mode } = await hybridIndex.hybridSearch( @@ -1443,7 +1630,7 @@ describe("hybridSearch — folder-scoped vector candidate window", () => { for (const fileNumber of [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]) { const filePath = `Archive/outside-${fileNumber}.txt` fileIndex.upsertNonMdFile(filePath, 100) - fileIndex.upsertFileContent( + const sourceVersion = fileIndex.upsertFileContent( { filePath, rawContent: "Unrelated archive material.", @@ -1451,10 +1638,10 @@ describe("hybridSearch — folder-scoped vector candidate window", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath }, logger) + await fileIndex.embedFileContent({ sourceVersion, filePath }, logger) } fileIndex.upsertNonMdFile("Docs/inside.txt", 100) - fileIndex.upsertFileContent( + const insideFileSourceVersion = fileIndex.upsertFileContent( { filePath: "Docs/inside.txt", rawContent: "The one file that lives inside Docs.", @@ -1462,7 +1649,10 @@ describe("hybridSearch — folder-scoped vector candidate window", () => { }, logger, ) - await fileIndex.embedFileContent({ filePath: "Docs/inside.txt" }, logger) + await fileIndex.embedFileContent( + { sourceVersion: insideFileSourceVersion, filePath: "Docs/inside.txt" }, + logger, + ) const { results, search_mode } = await fileIndex.hybridSearch( { query: "zzqq", filters: { folder: "Docs" }, limit: 1 }, @@ -1496,7 +1686,7 @@ describe("hybridSearch — ranking tuning", () => { * shape, where the file earns two leg ranks and the note only its * vector leg. */ const seedNoteAndFile = async (index: ReturnType): Promise => { - index.upsertNote( + const careerNoteSourceVersion = index.upsertNote( { filePath: "notes/career.md", rawContent: NOTE_CONTENT, @@ -1504,9 +1694,16 @@ describe("hybridSearch — ranking tuning", () => { }, logger, ) - await index.embedNote({ notePath: "notes/career.md", rawContent: NOTE_CONTENT }, logger) + await index.embedNote( + { + sourceVersion: careerNoteSourceVersion, + notePath: "notes/career.md", + rawContent: NOTE_CONTENT, + }, + logger, + ) index.upsertNonMdFile("docs/guide.txt", 200) - index.upsertFileContent( + const guideFileSourceVersion = index.upsertFileContent( { filePath: "docs/guide.txt", rawContent: FILE_CONTENT, @@ -1514,7 +1711,10 @@ describe("hybridSearch — ranking tuning", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/guide.txt" }, logger) + await index.embedFileContent( + { sourceVersion: guideFileSourceVersion, filePath: "docs/guide.txt" }, + logger, + ) } it("file-leg weight demotes a two-leg file hit below a note-vector hit", async () => { @@ -1693,7 +1893,7 @@ describe("hybridSearch — ranking tuning", () => { ranking: { rerankKindPrefix: true }, }) // Seed a note so the reranker fires (needs >= 2 candidates) - index.upsertNote( + const careerNoteSourceVersion = index.upsertNote( { filePath: "notes/career.md", rawContent: NOTE_CONTENT, @@ -1701,9 +1901,16 @@ describe("hybridSearch — ranking tuning", () => { }, logger, ) - await index.embedNote({ notePath: "notes/career.md", rawContent: NOTE_CONTENT }, logger) + await index.embedNote( + { + sourceVersion: careerNoteSourceVersion, + notePath: "notes/career.md", + rawContent: NOTE_CONTENT, + }, + logger, + ) index.upsertNonMdFile("docs/report.pdf", 100) - index.upsertFileContent( + const reportFileSourceVersion = index.upsertFileContent( { filePath: "docs/report.pdf", rawContent: pdfContent, @@ -1711,7 +1918,10 @@ describe("hybridSearch — ranking tuning", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/report.pdf" }, logger) + await index.embedFileContent( + { sourceVersion: reportFileSourceVersion, filePath: "docs/report.pdf" }, + logger, + ) await index.hybridSearch({ query: "quarterly earnings" }, logger) @@ -1747,7 +1957,7 @@ describe("hybridSearch — ranking tuning", () => { ranking: { rerankKindPrefix: true }, }) // Seed a note so the reranker fires (needs >= 2 candidates) - index.upsertNote( + const careerNoteSourceVersion = index.upsertNote( { filePath: "notes/career.md", rawContent: NOTE_CONTENT, @@ -1755,9 +1965,16 @@ describe("hybridSearch — ranking tuning", () => { }, logger, ) - await index.embedNote({ notePath: "notes/career.md", rawContent: NOTE_CONTENT }, logger) + await index.embedNote( + { + sourceVersion: careerNoteSourceVersion, + notePath: "notes/career.md", + rawContent: NOTE_CONTENT, + }, + logger, + ) index.upsertNonMdFile("Diagrams/infra.canvas", 300) - index.upsertFileContent( + const infraFileSourceVersion = index.upsertFileContent( { filePath: "Diagrams/infra.canvas", rawContent: canvasJson, @@ -1765,7 +1982,10 @@ describe("hybridSearch — ranking tuning", () => { }, logger, ) - await index.embedFileContent({ filePath: "Diagrams/infra.canvas" }, logger) + await index.embedFileContent( + { sourceVersion: infraFileSourceVersion, filePath: "Diagrams/infra.canvas" }, + logger, + ) await index.hybridSearch({ query: "infrastructure deployment topology" }, logger) diff --git a/src/vault-mcp/search/__tests__/memory-index.test.ts b/src/vault-mcp/search/__tests__/memory-index.test.ts index 32de08b4d..96fb39824 100644 --- a/src/vault-mcp/search/__tests__/memory-index.test.ts +++ b/src/vault-mcp/search/__tests__/memory-index.test.ts @@ -171,7 +171,7 @@ describe("memory entry indexing", () => { const { index, embedder } = await createInspectableMemoryIndex() if (embedder === undefined) throw new Error("embedder required") - index.upsertNote( + const sourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: OPINIONS_V1, @@ -179,7 +179,10 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { sourceVersion: sourceVersion, notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, + logger, + ) const embeddedEntryTexts = embedder.embedBatch.mock.calls.flatMap( (call: unknown[]) => call[0] as string[], ) @@ -194,7 +197,7 @@ describe("memory entry indexing", () => { const { index, embedder, inspect } = await createInspectableMemoryIndex() if (embedder === undefined) throw new Error("embedder required") - index.upsertNote( + const originalSourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: OPINIONS_V1, @@ -202,7 +205,14 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { + sourceVersion: originalSourceVersion, + notePath: "About Me/Opinions.md", + rawContent: OPINIONS_V1, + }, + logger, + ) embedder.embedBatch.mockClear() // Top-insert (the memory append default) shifts every later entry's @@ -211,7 +221,7 @@ describe("memory entry indexing", () => { "- **2026-07-02**: Wrap function bodies in braces.", "- **2026-07-11**: Newest opinion lands on top.\n- **2026-07-02**: Wrap function bodies in braces.", ) - index.upsertNote( + const updatedSourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: withTopAppend, @@ -219,7 +229,14 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: withTopAppend }, logger) + await index.embedNote( + { + sourceVersion: updatedSourceVersion, + notePath: "About Me/Opinions.md", + rawContent: withTopAppend, + }, + logger, + ) expect(totalTextsEmbedded(embedder)).toBe(1) expect( @@ -236,7 +253,7 @@ describe("memory entry indexing", () => { const { index, embedder, inspect } = await createInspectableMemoryIndex() if (embedder === undefined) throw new Error("embedder required") - index.upsertNote( + const originalSourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: OPINIONS_V1, @@ -244,14 +261,21 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { + sourceVersion: originalSourceVersion, + notePath: "About Me/Opinions.md", + rawContent: OPINIONS_V1, + }, + logger, + ) embedder.embedBatch.mockClear() const withEdit = OPINIONS_V1.replace( "- **2026-05-07**: Immutable over mutable.", "- **2026-05-07**: Immutable over mutable, always.", ) - index.upsertNote( + const updatedSourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: withEdit, @@ -259,7 +283,14 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: withEdit }, logger) + await index.embedNote( + { + sourceVersion: updatedSourceVersion, + notePath: "About Me/Opinions.md", + rawContent: withEdit, + }, + logger, + ) expect(totalTextsEmbedded(embedder)).toBe(1) // Still 3 rows and 3 vectors — the old row and its vector are gone, not @@ -276,7 +307,7 @@ describe("memory entry indexing", () => { const { index, embedder, inspect } = await createInspectableMemoryIndex() if (embedder === undefined) throw new Error("embedder required") - index.upsertNote( + const sourceVersion = index.upsertNote( { filePath: "About Me/Routines.md", rawContent: OPINIONS_V1, @@ -284,7 +315,10 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Routines.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { sourceVersion: sourceVersion, notePath: "About Me/Routines.md", rawContent: OPINIONS_V1 }, + logger, + ) // The prune target was present before (the trigger state is real). expect(selectEntryRows(inspect).some((row) => row.entry_date === "2026-05-07")).toBe(true) expect(countVectors(inspect)).toBe(3) @@ -306,7 +340,7 @@ describe("memory entry indexing", () => { it("removeNote clears the file's entry rows, FTS rows, and vectors", async () => { const { index, inspect } = await createInspectableMemoryIndex() - index.upsertNote( + const sourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: OPINIONS_V1, @@ -314,7 +348,10 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { sourceVersion: sourceVersion, notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, + logger, + ) expect(selectEntryRows(inspect)).toHaveLength(3) index.removeNote("About Me/Opinions.md") @@ -330,7 +367,7 @@ describe("memory entry indexing", () => { const { index, embedder, inspect } = await createInspectableMemoryIndex() if (embedder === undefined) throw new Error("embedder required") - index.upsertNote( + const originalSourceVersion = index.upsertNote( { filePath: "About Me/Opinions.md", rawContent: OPINIONS_V1, @@ -338,12 +375,19 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { + sourceVersion: originalSourceVersion, + notePath: "About Me/Opinions.md", + rawContent: OPINIONS_V1, + }, + logger, + ) embedder.embedBatch.mockClear() // The watcher delivers a rename as unlink + add. index.removeNote("About Me/Opinions.md") - index.upsertNote( + const updatedSourceVersion = index.upsertNote( { filePath: "About Me/Beliefs.md", rawContent: OPINIONS_V1, @@ -351,7 +395,14 @@ describe("memory entry indexing", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Beliefs.md", rawContent: OPINIONS_V1 }, logger) + await index.embedNote( + { + sourceVersion: updatedSourceVersion, + notePath: "About Me/Beliefs.md", + rawContent: OPINIONS_V1, + }, + logger, + ) const rows = selectEntryRows(inspect) expect(rows).toHaveLength(3) @@ -407,6 +458,100 @@ describe("memory entry indexing", () => { }) }) +describe("memory embedding source versions", () => { + it.each(["delete", "recreate", "prune"] as const)( + "stops an obsolete 17-entry batch after a source %s", + async (mutation) => { + const { index, embedder, inspect } = await createInspectableMemoryIndex() + + if (!embedder) throw new Error("embedder required") + const content = `# Opinions\n\n## Practices\n\n${Array.from( + { length: 17 }, + (_, entryIndex) => { + return `- **2026-07-01**: Practice ${String(entryIndex)} improves reliability.` + }, + ).join("\n")}` + const firstBatchStarted = Promise.withResolvers() + const batchModel = Promise.withResolvers() + embedder.embedBatch.mockImplementationOnce(() => { + firstBatchStarted.resolve(undefined) + return batchModel.promise + }) + const debugSpy = vi.spyOn(logger, "debug").mockImplementation(() => {}) + onTestFinished(() => debugSpy.mockRestore()) + const originalVersion = index.upsertNote( + { filePath: "About Me/Opinions.md", rawContent: content, fileStat: testStat(1000) }, + logger, + ) + const originalIds = inspect + .prepare<[], { id: number }>("SELECT id FROM memory_entries ORDER BY id") + .all() + .map((row) => row.id) + expect(originalIds).toHaveLength(17) + const staleJob = index.embedNote( + { notePath: "About Me/Opinions.md", rawContent: content, sourceVersion: originalVersion }, + logger, + ) + await firstBatchStarted.promise + + const replacementContent = + mutation === "prune" + ? "# Opinions\n\n## Practices\n\n- **2026-07-01**: Current replacement practice." + : content + + if (mutation !== "prune") index.removeNote("About Me/Opinions.md") + const replacementVersion = + mutation === "delete" + ? null + : index.upsertNote( + { + filePath: "About Me/Opinions.md", + rawContent: replacementContent, + fileStat: testStat(1000), + }, + logger, + ) + batchModel.resolve(Array.from({ length: 16 }, () => new Float32Array(384).fill(0.1))) + await staleJob + + // Check the stale batch before replacement work can repair its state. + expect(embedder.embedBatch).toHaveBeenCalledTimes(1) + expect(countVectors(inspect)).toBe(0) + const expectedEntryCounts = { delete: 0, prune: 1, recreate: 17 } + expect(selectEntryRows(inspect)).toHaveLength(expectedEntryCounts[mutation]) + expect(debugSpy).toHaveBeenCalledWith("skipped obsolete embedding", { + path: "About Me/Opinions.md", + }) + const replacementIds = inspect + .prepare<[], { id: number }>("SELECT id FROM memory_entries ORDER BY id") + .all() + .map((row) => row.id) + expect(replacementIds.filter((entryId) => originalIds.includes(entryId))).toEqual([]) + + if (replacementVersion === null) return + embedder.embedBatch.mockClear() + await index.embedNote( + { + notePath: "About Me/Opinions.md", + rawContent: replacementContent, + sourceVersion: replacementVersion, + }, + logger, + ) + expect(embedder.embedBatch).toHaveBeenCalledTimes(mutation === "prune" ? 1 : 2) + expect(countVectors(inspect)).toBe(replacementIds.length) + expect( + inspect + .prepare<[], { entry_id: number }>( + "SELECT entry_id FROM memory_entry_vectors ORDER BY entry_id", + ) + .all() + .map((row) => row.entry_id), + ).toEqual(replacementIds) + }, + ) +}) + describe("memory entry rebuild reconciliation", () => { const AGENTS_MD = `--- title: Agents @@ -477,6 +622,10 @@ title: Agents await firstBuild.embedding const rowsAfterFirstBuild = selectEntryRows(inspect) expect(rowsAfterFirstBuild).toHaveLength(4) + const chunksAfterFirstBuild = inspect + .prepare("SELECT id, note_path, chunk_text, content_hash FROM note_chunks ORDER BY id") + .all() + expect(chunksAfterFirstBuild).toHaveLength(2) const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}) onTestFinished(() => warnSpy.mockRestore()) @@ -493,6 +642,12 @@ title: Agents // A still-on-disk memory note must not be treated as deleted — its // entry rows survive until a rebuild parses it successfully again expect(selectEntryRows(inspect)).toEqual(rowsAfterFirstBuild) + expect( + inspect + .prepare("SELECT id, note_path, chunk_text, content_hash FROM note_chunks ORDER BY id") + .all(), + ).toEqual(chunksAfterFirstBuild) + expect(countVectors(inspect)).toBe(4) }) it("embeds zero entries on a second rebuild with unchanged files", async () => { diff --git a/src/vault-mcp/search/__tests__/memory-recall.test.ts b/src/vault-mcp/search/__tests__/memory-recall.test.ts index 78ec3eaac..3ac0806ff 100644 --- a/src/vault-mcp/search/__tests__/memory-recall.test.ts +++ b/src/vault-mcp/search/__tests__/memory-recall.test.ts @@ -82,11 +82,11 @@ const createRecallIndex = async (options?: { const files = options?.files ?? DEFAULT_FILES for (const [fileName, content] of Object.entries(files)) { const filePath = `About Me/${fileName}.md` - index.upsertNote( + const sourceVersion = index.upsertNote( { filePath, rawContent: content, fileStat: { mtimeMs: 1000, size: 100 } }, logger, ) - await index.embedNote({ notePath: filePath, rawContent: content }, logger) + await index.embedNote({ sourceVersion, notePath: filePath, rawContent: content }, logger) } return index } @@ -334,7 +334,7 @@ describe("memoryRecall", () => { // Seed both files so we have vector-only AND lexical entries for (const [fileName, content] of Object.entries(DEFAULT_FILES)) { const filePath = `About Me/${fileName}.md` - index.upsertNote( + const sourceVersion = index.upsertNote( { filePath, rawContent: content, @@ -342,7 +342,7 @@ describe("memoryRecall", () => { }, logger, ) - await index.embedNote({ notePath: filePath, rawContent: content }, logger) + await index.embedNote({ sourceVersion, notePath: filePath, rawContent: content }, logger) } // Break the embedder for the recall query — memoryVectorSearch catches @@ -497,7 +497,7 @@ describe("memoryRecall", () => { for (const [fileName, topicMarker] of seededFiles) { const filePath = `About Me/${fileName}.md` const content = `# ${fileName}\n\n## Working style (newest first)\n\n- **2026-07-02**: Pacing beats crunch on ${topicMarker}.\n` - index.upsertNote( + const sourceVersion = index.upsertNote( { filePath, rawContent: content, @@ -505,7 +505,7 @@ describe("memoryRecall", () => { }, logger, ) - await index.embedNote({ notePath: filePath, rawContent: content }, logger) + await index.embedNote({ sourceVersion, notePath: filePath, rawContent: content }, logger) } const result = await index.memoryRecall({ query: "pacing crunch", limit: 1 }, logger) @@ -546,7 +546,7 @@ describe("memoryRecall", () => { (_, fillerIndex) => `- **2026-07-02**: Background logistics note ${String(fillerIndex)}.`, ).join("\n") const content = `# Ledger\n\n## Working style (newest first)\n\n${fillerEntries}\n- **2026-07-02**: Pacing beats crunch on alpha-topic.\n- **2026-07-02**: Pacing beats crunch on beta-topic.\n` - index.upsertNote( + const ledgerNoteSourceVersion = index.upsertNote( { filePath: "About Me/Ledger.md", rawContent: content, @@ -554,7 +554,14 @@ describe("memoryRecall", () => { }, logger, ) - await index.embedNote({ notePath: "About Me/Ledger.md", rawContent: content }, logger) + await index.embedNote( + { + sourceVersion: ledgerNoteSourceVersion, + notePath: "About Me/Ledger.md", + rawContent: content, + }, + logger, + ) const result = await index.memoryRecall({ query: "pacing crunch", limit: 1 }, logger) expect(result.entries.map((entry) => entry.text)).toEqual([ diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index 740b379c2..b67784818 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -63,6 +63,84 @@ const countRow = (row: unknown): { count: number } => { throw new Error("expected a count row") } +const seedEmbeddingSource = ( + searchIndex: SearchIndex, + params: { notePath: string; rawContent: string }, +): symbol => { + return searchIndex.upsertNote( + { + filePath: params.notePath, + rawContent: params.rawContent, + fileStat: { mtimeMs: 1000, size: Buffer.byteLength(params.rawContent) }, + }, + logger, + ) +} + +const createEmbeddingRaceIndex = async (sourceKind: "note" | "file") => { + const dir = await mkdtemp(join(tmpdir(), "embedding-race-")) + onTestFinished(() => rm(dir, { recursive: true, force: true })) + const embedder = { + embedText: vi.fn().mockResolvedValue(new Float32Array(384).fill(0.1)), + embedBatch: vi.fn().mockImplementation((texts: string[]) => { + return Promise.resolve(texts.map(() => new Float32Array(384).fill(0.1))) + }), + } + const dbPath = join(dir, "index.db") + const searchIndex = createSearchIndex(dbPath, embedder, undefined, { fileToolsEnabled: true }) + const inspect = new Database(dbPath) + sqliteVec.load(inspect) + onTestFinished(() => { + inspect.close() + }) + const path = sourceKind === "note" ? "reuse.md" : "reuse.txt" + const chunkTable = sourceKind === "note" ? "note_chunks" : "file_content_chunks" + const vectorTable = sourceKind === "note" ? "note_vectors" : "file_content_vectors" + const sourceTable = sourceKind === "note" ? "notes" : "file_content" + const upsert = (content: string): symbol => { + const params = { filePath: path, rawContent: content, fileStat: testStat(1000) } + + return sourceKind === "note" + ? searchIndex.upsertNote(params, logger) + : searchIndex.upsertFileContent(params, logger) + } + const embed = (content: string, sourceVersion: symbol): Promise => { + return sourceKind === "note" + ? searchIndex.embedNote({ notePath: path, rawContent: content, sourceVersion }, logger) + : searchIndex.embedFileContent({ filePath: path, sourceVersion }, logger) + } + const remove = (): void => { + if (sourceKind === "note") { + searchIndex.removeNote(path) + return + } + searchIndex.removeFileContent({ filePath: path }, logger) + } + const chunks = (): Array<{ chunk_index: number; chunk_text: string }> => { + return inspect + .prepare<[], { chunk_index: number; chunk_text: string }>( + `SELECT chunk_index, chunk_text FROM ${chunkTable} ORDER BY chunk_index`, + ) + .all() + } + const vectorCount = (): number => { + return countRow(inspect.prepare(`SELECT COUNT(*) AS count FROM ${vectorTable}`).get()).count + } + return { + searchIndex, + embedder, + inspect, + dir, + path, + sourceTable, + upsert, + embed, + remove, + chunks, + vectorCount, + } +} + /** Builds a fileStat object for upsertNote. Defaults to size 100. */ const testStat = (mtimeMs: number, size = 100): { mtimeMs: number; size: number } => ({ mtimeMs, @@ -483,7 +561,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { // reverse, so the asserted order can come only from the secondary sort // keys — whichever way a vec0 build returns tied distances. for (const notePath of ["mmm.md", "zzz.md", "aaa.md"]) { - tieIndex.upsertNote( + const sourceVersion = tieIndex.upsertNote( { filePath: notePath, rawContent: IDENTICAL_NOTE, @@ -491,7 +569,10 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedNote({ notePath, rawContent: IDENTICAL_NOTE }, logger) + await tieIndex.embedNote( + { sourceVersion: sourceVersion, notePath, rawContent: IDENTICAL_NOTE }, + logger, + ) } // "orca" shares no stems with the note content, so the FTS leg is empty @@ -510,7 +591,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { // keys — whichever way a vec0 build returns tied distances. for (const filePath of ["mmm.txt", "zzz.txt", "aaa.txt"]) { tieIndex.upsertNonMdFile(filePath, 100) - tieIndex.upsertFileContent( + const sourceVersion = tieIndex.upsertFileContent( { filePath, rawContent: identicalFileContent, @@ -518,7 +599,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedFileContent({ filePath }, logger) + await tieIndex.embedFileContent({ sourceVersion: sourceVersion, filePath }, logger) } // "orca" shares no stems with the file content, so the FTS legs are @@ -536,7 +617,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { // post-SQL note filter under either KNN statement — this pins the // in-folder statement's ordering keys, not which statement ran. for (const notePath of ["docs/mmm.md", "docs/zzz.md", "other/out.md", "docs/aaa.md"]) { - tieIndex.upsertNote( + const sourceVersion = tieIndex.upsertNote( { filePath: notePath, rawContent: IDENTICAL_NOTE, @@ -544,7 +625,10 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedNote({ notePath, rawContent: IDENTICAL_NOTE }, logger) + await tieIndex.embedNote( + { sourceVersion: sourceVersion, notePath, rawContent: IDENTICAL_NOTE }, + logger, + ) } const { results } = await tieIndex.hybridSearch( @@ -568,7 +652,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { // in-folder statement ran. for (const filePath of ["docs/mmm.txt", "docs/zzz.txt", "other/out.txt", "docs/aaa.txt"]) { tieIndex.upsertNonMdFile(filePath, 100) - tieIndex.upsertFileContent( + const sourceVersion = tieIndex.upsertFileContent( { filePath, rawContent: identicalFileContent, @@ -576,7 +660,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedFileContent({ filePath }, logger) + await tieIndex.embedFileContent({ sourceVersion: sourceVersion, filePath }, logger) } const { results } = await tieIndex.hybridSearch( @@ -607,7 +691,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { "ggg.md", "hhh.md", ]) { - tieIndex.upsertNote( + const sourceVersion = tieIndex.upsertNote( { filePath: notePath, rawContent: IDENTICAL_NOTE, @@ -615,7 +699,10 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedNote({ notePath, rawContent: IDENTICAL_NOTE }, logger) + await tieIndex.embedNote( + { sourceVersion: sourceVersion, notePath, rawContent: IDENTICAL_NOTE }, + logger, + ) } const { results } = await tieIndex.hybridSearch({ query: "orca", limit: 2 }, logger) @@ -639,7 +726,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { "hhh.txt", ]) { tieIndex.upsertNonMdFile(filePath, 100) - tieIndex.upsertFileContent( + const sourceVersion = tieIndex.upsertFileContent( { filePath, rawContent: identicalFileContent, @@ -647,7 +734,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedFileContent({ filePath }, logger) + await tieIndex.embedFileContent({ sourceVersion: sourceVersion, filePath }, logger) } const { results } = await tieIndex.hybridSearch({ query: "orca", limit: 2 }, logger) @@ -672,7 +759,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { "docs/ggg.md", "docs/hhh.md", ]) { - tieIndex.upsertNote( + const sourceVersion = tieIndex.upsertNote( { filePath: notePath, rawContent: IDENTICAL_NOTE, @@ -680,7 +767,10 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedNote({ notePath, rawContent: IDENTICAL_NOTE }, logger) + await tieIndex.embedNote( + { sourceVersion: sourceVersion, notePath, rawContent: IDENTICAL_NOTE }, + logger, + ) } const { results } = await tieIndex.hybridSearch( @@ -712,7 +802,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { "docs/hhh.txt", ]) { tieIndex.upsertNonMdFile(filePath, 100) - tieIndex.upsertFileContent( + const sourceVersion = tieIndex.upsertFileContent( { filePath, rawContent: identicalFileContent, @@ -720,7 +810,7 @@ describe("equal-score tie-breaking in retrieval legs", () => { }, logger, ) - await tieIndex.embedFileContent({ filePath }, logger) + await tieIndex.embedFileContent({ sourceVersion: sourceVersion, filePath }, logger) } const { results } = await tieIndex.hybridSearch( @@ -5042,8 +5132,13 @@ It has multiple sentences to verify chunking works correctly. const mockEmbedder = createMockEmbedder() const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) + const sourceVersion = seedEmbeddingSource(embeddingIndex, { + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }) + await embeddingIndex.embedNote( - { notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + { sourceVersion: sourceVersion, notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, logger, ) @@ -5056,8 +5151,15 @@ It has multiple sentences to verify chunking works correctly. ranking: { enrichChunkMetadata: true }, }) + const sourceVersion = seedEmbeddingSource(enrichedIndex, { + notePath: "typed.md", + rawContent: + "---\ntitle: Typed Note\ntype: reference\ntags: [search, ranking]\n---\n\nBody content for enrichment.\n", + }) + await enrichedIndex.embedNote( { + sourceVersion: sourceVersion, notePath: "typed.md", rawContent: "---\ntitle: Typed Note\ntype: reference\ntags: [search, ranking]\n---\n\nBody content for enrichment.\n", @@ -5077,8 +5179,14 @@ It has multiple sentences to verify chunking works correctly. ranking: { enrichChunkMetadata: true }, }) + const sourceVersion = seedEmbeddingSource(enrichedIndex, { + notePath: "bare.md", + rawContent: "---\ntitle: Bare Note\n---\n\nBody without metadata.\n", + }) + await enrichedIndex.embedNote( { + sourceVersion: sourceVersion, notePath: "bare.md", rawContent: "---\ntitle: Bare Note\n---\n\nBody without metadata.\n", }, @@ -5093,8 +5201,15 @@ It has multiple sentences to verify chunking works correctly. const mockEmbedder = createMockEmbedder() const defaultIndex = createSearchIndex(":memory:", mockEmbedder) + const sourceVersion = seedEmbeddingSource(defaultIndex, { + notePath: "typed.md", + rawContent: + "---\ntitle: Typed Note\ntype: reference\ntags: [search, ranking]\n---\n\nBody content for enrichment.\n", + }) + await defaultIndex.embedNote( { + sourceVersion: sourceVersion, notePath: "typed.md", rawContent: "---\ntitle: Typed Note\ntype: reference\ntags: [search, ranking]\n---\n\nBody content for enrichment.\n", @@ -5113,15 +5228,20 @@ It has multiple sentences to verify chunking works correctly. const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) // First embed + const sourceVersion = seedEmbeddingSource(embeddingIndex, { + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }) + await embeddingIndex.embedNote( - { notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + { sourceVersion: sourceVersion, notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, logger, ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) // Second embed with same content — should skip (hash match) await embeddingIndex.embedNote( - { notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + { sourceVersion: sourceVersion, notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, logger, ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) @@ -5131,8 +5251,17 @@ It has multiple sentences to verify chunking works correctly. const mockEmbedder = createMockEmbedder() const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) + const originalSourceVersion = seedEmbeddingSource(embeddingIndex, { + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }) + await embeddingIndex.embedNote( - { notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + { + sourceVersion: originalSourceVersion, + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }, logger, ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) @@ -5141,7 +5270,15 @@ It has multiple sentences to verify chunking works correctly. "multiple sentences", "different content entirely", ) - await embeddingIndex.embedNote({ notePath: "test.md", rawContent: updatedNote }, logger) + const updatedSourceVersion = seedEmbeddingSource(embeddingIndex, { + notePath: "test.md", + rawContent: updatedNote, + }) + + await embeddingIndex.embedNote( + { sourceVersion: updatedSourceVersion, notePath: "test.md", rawContent: updatedNote }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(2) }) @@ -5149,7 +5286,7 @@ It has multiple sentences to verify chunking works correctly. const mockEmbedder = createMockEmbedder() const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) - embeddingIndex.upsertNote( + const originalSourceVersion = embeddingIndex.upsertNote( { filePath: "test.md", rawContent: NOTE_FOR_EMBEDDING, @@ -5158,7 +5295,11 @@ It has multiple sentences to verify chunking works correctly. logger, ) await embeddingIndex.embedNote( - { notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + { + sourceVersion: originalSourceVersion, + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }, logger, ) @@ -5167,7 +5308,7 @@ It has multiple sentences to verify chunking works correctly. // Re-embedding after removal should embed again (not skip via hash) mockEmbedder.embedText.mockClear() - embeddingIndex.upsertNote( + const updatedSourceVersion = embeddingIndex.upsertNote( { filePath: "test.md", rawContent: NOTE_FOR_EMBEDDING, @@ -5176,7 +5317,11 @@ It has multiple sentences to verify chunking works correctly. logger, ) await embeddingIndex.embedNote( - { notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + { + sourceVersion: updatedSourceVersion, + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }, logger, ) expect(mockEmbedder.embedText).toHaveBeenCalled() @@ -5186,7 +5331,15 @@ It has multiple sentences to verify chunking works correctly. const mockEmbedder = createMockEmbedder() const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) - await embeddingIndex.embedNote({ notePath: "empty.md", rawContent: "" }, logger) + const sourceVersion = seedEmbeddingSource(embeddingIndex, { + notePath: "empty.md", + rawContent: "", + }) + + await embeddingIndex.embedNote( + { sourceVersion: sourceVersion, notePath: "empty.md", rawContent: "" }, + logger, + ) // chunker returns at least one chunk (the title-only fallback), so // embedText is called even for empty content @@ -5198,8 +5351,16 @@ It has multiple sentences to verify chunking works correctly. mockEmbedder.embedText.mockRejectedValueOnce(new Error("embedding failed")) const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) + const sourceVersion = seedEmbeddingSource(embeddingIndex, { + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }) + await expect( - embeddingIndex.embedNote({ notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, logger), + embeddingIndex.embedNote( + { sourceVersion: sourceVersion, notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + logger, + ), ).rejects.toThrow("embedding failed") }) }) @@ -5208,8 +5369,16 @@ It has multiple sentences to verify chunking works correctly. it("embedNote is a no-op when no embedder is provided", async () => { const noEmbedIndex = createSearchIndex(":memory:") + const sourceVersion = seedEmbeddingSource(noEmbedIndex, { + notePath: "test.md", + rawContent: NOTE_FOR_EMBEDDING, + }) + await expect( - noEmbedIndex.embedNote({ notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, logger), + noEmbedIndex.embedNote( + { sourceVersion: sourceVersion, notePath: "test.md", rawContent: NOTE_FOR_EMBEDDING }, + logger, + ), ).resolves.toBeUndefined() }) @@ -6175,6 +6344,419 @@ describe("canvas file content and links", () => { // ── File content vector embeddings ──────────────────────────── +describe("committed embedding source versions", () => { + it.each(["note", "file"] as const)( + "rejects a %s model result after source deletion", + async (sourceKind) => { + const fixture = await createEmbeddingRaceIndex(sourceKind) + const model = Promise.withResolvers() + fixture.embedder.embedText.mockImplementationOnce(() => model.promise) + const debugSpy = vi.spyOn(logger, "debug").mockImplementation(() => {}) + onTestFinished(() => debugSpy.mockRestore()) + const sourceVersion = fixture.upsert("Old oldquartz") + const job = fixture.embed("Old oldquartz", sourceVersion) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(1) + + fixture.remove() + expect(fixture.inspect.prepare(`SELECT path FROM ${fixture.sourceTable}`).all()).toEqual([]) + model.resolve(new Float32Array(384).fill(0.1)) + await job + + expect(fixture.chunks()).toEqual([]) + expect(fixture.vectorCount()).toBe(0) + expect(debugSpy).toHaveBeenCalledWith("skipped obsolete embedding", { path: fixture.path }) + }, + ) + + it.each(["note", "file"] as const)( + "keeps a recreated %s free of stale chunks while its replacement model is held", + async (sourceKind) => { + const fixture = await createEmbeddingRaceIndex(sourceKind) + const oldModel = Promise.withResolvers() + const replacementModel = Promise.withResolvers() + fixture.embedder.embedText.mockImplementationOnce(() => oldModel.promise) + fixture.embedder.embedText.mockImplementationOnce(() => replacementModel.promise) + const originalContent = + sourceKind === "note" ? "---\ntitle: Old\n---\nOld oldquartz" : "Old oldquartz" + const replacementContent = + sourceKind === "note" ? "---\ntitle: New\n---\nNew newcobalt" : "New newcobalt" + const replacementTitle = sourceKind === "note" ? "New" : "reuse" + const originalVersion = fixture.upsert(originalContent) + const oldJob = fixture.embed(originalContent, originalVersion) + fixture.remove() + const replacementVersion = fixture.upsert(replacementContent) + const replacementJob = fixture.embed(replacementContent, replacementVersion) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(2) + + oldModel.resolve(new Float32Array(384).fill(0.1)) + await oldJob + expect( + fixture.inspect.prepare(`SELECT title, content FROM ${fixture.sourceTable}`).all(), + ).toEqual([{ title: replacementTitle, content: "New newcobalt" }]) + expect(fixture.chunks()).toEqual([]) + expect(fixture.vectorCount()).toBe(0) + + replacementModel.resolve(new Float32Array(384).fill(0.1)) + await replacementJob + expect(fixture.chunks()).toEqual([ + { chunk_index: 0, chunk_text: `${replacementTitle}\n\nNew newcobalt` }, + ]) + expect(fixture.vectorCount()).toBe(1) + const { results } = await fixture.searchIndex.hybridSearch( + { query: "unmatchedsemanticquery" }, + logger, + ) + expect( + results.map((result) => ({ + path: result.path, + title: result.title, + snippet: result.snippet, + })), + ).toEqual([ + { + path: fixture.path, + title: replacementTitle, + snippet: `${replacementTitle} New newcobalt`, + }, + ]) + }, + ) + + it.each(["note", "file"] as const)( + "does not let a stale short %s job prune a newer long source's tail", + async (sourceKind) => { + const fixture = await createEmbeddingRaceIndex(sourceKind) + const staleModel = Promise.withResolvers() + const shortVersion = fixture.upsert("Short obsolete body.") + fixture.embedder.embedText.mockImplementationOnce(() => staleModel.promise) + const staleJob = fixture.embed("Short obsolete body.", shortVersion) + const longContent = Array.from({ length: 8 }, (_, paragraphIndex) => { + return `## Section ${String(paragraphIndex)}\n\n${"Current content about cobalt systems. ".repeat(25)}` + }).join("\n\n") + const currentVersion = fixture.upsert(longContent) + await fixture.embed(longContent, currentVersion) + const currentChunks = fixture.chunks() + expect(currentChunks.length).toBeGreaterThan(1) + expect(fixture.vectorCount()).toBe(currentChunks.length) + + staleModel.resolve(new Float32Array(384).fill(0.1)) + await staleJob + expect(fixture.chunks()).toEqual(currentChunks) + expect(fixture.vectorCount()).toBe(currentChunks.length) + }, + ) + + it.each(["note", "file"] as const)( + "preserves a %s source version when its upsert transaction fails", + async (sourceKind) => { + const sqlFragment = + sourceKind === "note" ? "INSERT INTO tasks" : "INSERT INTO file_content_fts" + const poison = installStatementPoison(sqlFragment) + const fixture = await createEmbeddingRaceIndex(sourceKind) + const originalContent = "Original committed content." + const sourceVersion = fixture.upsert(originalContent) + poison.arm() + expect(() => fixture.upsert("Replacement body.\n\n- [ ] trigger task")).toThrow( + poison.message, + ) + poison.disarm() + + await fixture.embed(originalContent, sourceVersion) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(1) + expect(fixture.chunks()).toEqual([ + { chunk_index: 0, chunk_text: "reuse\n\nOriginal committed content." }, + ]) + expect(fixture.inspect.prepare(`SELECT content FROM ${fixture.sourceTable}`).all()).toEqual([ + { content: originalContent }, + ]) + }, + ) + + it.each(["note", "file"] as const)( + "preserves a %s source version when its removal transaction fails", + async (sourceKind) => { + const sqlFragment = + sourceKind === "note" ? "DELETE FROM tasks" : "DELETE FROM file_content WHERE" + const poison = installStatementPoison(sqlFragment) + const fixture = await createEmbeddingRaceIndex(sourceKind) + const content = "Original retained source." + const sourceVersion = fixture.upsert(content) + poison.arm() + expect(fixture.remove).toThrow(poison.message) + poison.disarm() + + await fixture.embed(content, sourceVersion) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(1) + expect(fixture.chunks()).toEqual([ + { chunk_index: 0, chunk_text: "reuse\n\nOriginal retained source." }, + ]) + expect(fixture.inspect.prepare(`SELECT content FROM ${fixture.sourceTable}`).all()).toEqual([ + { content }, + ]) + }, + ) + + it("preserves the committed note version when replacement frontmatter cannot parse", async () => { + const fixture = await createEmbeddingRaceIndex("note") + const content = "Original retained source." + const sourceVersion = fixture.upsert(content) + expect(() => fixture.upsert("---\ntitle: [unclosed\n---\nReplacement.")).toThrow( + "Flow sequence in block collection must be sufficiently indented and end with a ]", + ) + + await fixture.embed(content, sourceVersion) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(1) + expect(fixture.chunks()).toEqual([ + { chunk_index: 0, chunk_text: "reuse\n\nOriginal retained source." }, + ]) + expect(fixture.inspect.prepare("SELECT content FROM notes").all()).toEqual([{ content }]) + }) + + it("skips deleted and same-mtime superseded rebuild snapshots before later model work", async () => { + const fixture = await createEmbeddingRaceIndex("note") + const infoSpy = vi.spyOn(logger, "info").mockImplementation(() => {}) + onTestFinished(() => infoSpy.mockRestore()) + const vaultPath = join(fixture.dir, "vault") + await mkdir(vaultPath) + await writeFile(join(vaultPath, "a.md"), "Old blocking snapshot.") + await writeFile(join(vaultPath, "b.md"), "Old queued snapshot.") + await writeFile(join(vaultPath, "queued.txt"), "Old file snapshot.") + const model = Promise.withResolvers() + fixture.embedder.embedText.mockImplementationOnce(() => model.promise) + const { embedding } = await fixture.searchIndex.rebuildFromVault({ vaultPath }, logger) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(1) + const secondMtime = fixture.inspect + .prepare<[], { mtime: number }>("SELECT mtime FROM notes WHERE path = 'b.md'") + .get()?.mtime + const fileMtime = fixture.inspect + .prepare<[], { mtime: number }>("SELECT mtime FROM file_content WHERE path = 'queued.txt'") + .get()?.mtime + + if (secondMtime === undefined || fileMtime === undefined) + throw new Error("rebuild sources missing") + fixture.searchIndex.removeNote("a.md") + const noteVersion = fixture.searchIndex.upsertNote( + { filePath: "b.md", rawContent: "New current note.", fileStat: testStat(secondMtime) }, + logger, + ) + const fileVersion = fixture.searchIndex.upsertFileContent( + { filePath: "queued.txt", rawContent: "New current file.", fileStat: testStat(fileMtime) }, + logger, + ) + await fixture.searchIndex.embedNote( + { notePath: "b.md", rawContent: "New current note.", sourceVersion: noteVersion }, + logger, + ) + await fixture.searchIndex.embedFileContent( + { filePath: "queued.txt", sourceVersion: fileVersion }, + logger, + ) + model.resolve(new Float32Array(384).fill(0.1)) + await embedding + + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(3) + expect( + fixture.inspect + .prepare("SELECT note_path, chunk_text FROM note_chunks ORDER BY note_path") + .all(), + ).toEqual([{ note_path: "b.md", chunk_text: "b\n\nNew current note." }]) + expect( + fixture.inspect.prepare("SELECT file_path, chunk_text FROM file_content_chunks").all(), + ).toEqual([{ file_path: "queued.txt", chunk_text: "queued\n\nNew current file." }]) + expect(fixture.vectorCount()).toBe(1) + expect(infoSpy).toHaveBeenCalledWith("embedding pass complete", { notes: 2, chunksEmbedded: 0 }) + expect(infoSpy).toHaveBeenCalledWith("file content embedding pass complete", { + files: 1, + fileChunksEmbedded: 0, + }) + }) + + it("rejects an outer rebuild rollback after a nested upsert without launching models and recovers", async () => { + const poison = installStatementPoison("DELETE FROM memory_entries WHERE file = ?") + const dir = await mkdtemp(join(tmpdir(), "embedding-rebuild-rollback-")) + onTestFinished(() => rm(dir, { recursive: true, force: true })) + const vaultPath = join(dir, "vault") + await mkdir(vaultPath) + await writeFile(join(vaultPath, "new.md"), "New successfully parsed source.") + const embedder = { + embedText: vi.fn().mockResolvedValue(new Float32Array(384).fill(0.1)), + embedBatch: vi.fn().mockResolvedValue([]), + } + const index = createSearchIndex(join(dir, "index.db"), embedder, undefined, { + memoryDir: "About Me", + }) + index.upsertNote( + { + filePath: "About Me/Old.md", + rawContent: "## Practices\n\n- **2026-07-01**: Old memory entry.", + fileStat: testStat(1000), + }, + logger, + ) + const inspect = new Database(join(dir, "index.db"), { readonly: true }) + sqliteVec.load(inspect) + onTestFinished(() => { + inspect.close() + }) + const debugSpy = vi.spyOn(logger, "debug").mockImplementation(() => {}) + onTestFinished(() => debugSpy.mockRestore()) + poison.arm() + + await expect(index.rebuildFromVault({ vaultPath }, logger)).rejects.toThrow(poison.message) + expect(debugSpy).toHaveBeenCalledWith("indexed note", { + path: "new.md", + bytes: 31, + tasksIndexed: 0, + }) + expect(inspect.prepare("SELECT path FROM notes").all()).toEqual([]) + expect(inspect.prepare("SELECT entry_text FROM memory_entries").all()).toEqual([ + { entry_text: "- **2026-07-01**: Old memory entry." }, + ]) + expect(embedder.embedText).not.toHaveBeenCalled() + expect(embedder.embedBatch).not.toHaveBeenCalled() + + poison.disarm() + const recovered = await index.rebuildFromVault({ vaultPath }, logger) + await recovered.embedding + expect(recovered.count).toBe(1) + expect(embedder.embedText).toHaveBeenCalledTimes(1) + expect(inspect.prepare("SELECT note_path, chunk_text FROM note_chunks").all()).toEqual([ + { note_path: "new.md", chunk_text: "new\n\nNew successfully parsed source." }, + ]) + expect(inspect.prepare("SELECT entry_text FROM memory_entries").all()).toEqual([]) + }) + + it("removes only parentless vectors on startup and restores nearest-neighbor capacity idempotently", async () => { + const dir = await mkdtemp(join(tmpdir(), "embedding-orphan-sweep-")) + onTestFinished(() => rm(dir, { recursive: true, force: true })) + const vaultPath = join(dir, "vault") + await mkdir(join(vaultPath, "About Me"), { recursive: true }) + await writeFile(join(vaultPath, "note.md"), "Current note body.") + await writeFile(join(vaultPath, "guide.txt"), "Current file body.") + await writeFile( + join(vaultPath, "About Me/Practices.md"), + "## Practices\n\n- **2026-07-01**: Keep current entries.", + ) + const embedder = { + embedText: vi.fn().mockResolvedValue(new Float32Array(384).fill(0.1)), + embedBatch: vi + .fn() + .mockImplementation((texts: string[]) => + Promise.resolve(texts.map(() => new Float32Array(384).fill(0.1))), + ), + } + const dbPath = join(dir, "index.db") + const index = createSearchIndex(dbPath, embedder, undefined, { + fileToolsEnabled: true, + memoryDir: "About Me", + }) + const initial = await index.rebuildFromVault({ vaultPath }, logger) + await initial.embedding + const inspect = new Database(dbPath) + sqliteVec.load(inspect) + onTestFinished(() => { + inspect.close() + }) + const queryVector = new Float32Array(384) + queryVector[0] = 1 + const queryBytes = Buffer.from(queryVector.buffer) + const stores = [ + { + vectorTable: "note_vectors", + parentTable: "note_chunks", + vectorKey: "chunk_id", + expectedParents: 2, + }, + { + vectorTable: "file_content_vectors", + parentTable: "file_content_chunks", + vectorKey: "chunk_id", + expectedParents: 1, + }, + { + vectorTable: "memory_entry_vectors", + parentTable: "memory_entries", + vectorKey: "entry_id", + expectedParents: 1, + }, + ] + const retainedStores = stores.map((store) => { + const parents = inspect.prepare(`SELECT * FROM ${store.parentTable} ORDER BY id`).all() + expect(parents).toHaveLength(store.expectedParents) + const vectors = inspect + .prepare( + `SELECT ${store.vectorKey}, hex(embedding) AS embedding FROM ${store.vectorTable} ORDER BY ${store.vectorKey}`, + ) + .all() + const nearestParents = inspect.prepare<[Buffer, number], { id: number }>( + `SELECT parent.id FROM ${store.vectorTable} vector JOIN ${store.parentTable} parent ON parent.id = vector.${store.vectorKey} + WHERE vector.embedding MATCH ? AND vector.k = ? ORDER BY vector.distance, parent.id`, + ) + const expectedHits = nearestParents.all(queryBytes, 2) + expect(expectedHits).toHaveLength(store.expectedParents) + const insertOrphan = inspect.prepare( + `INSERT INTO ${store.vectorTable} (${store.vectorKey}, embedding) VALUES (?, ?)`, + ) + insertOrphan.run(10000n, queryBytes) + insertOrphan.run(10001n, queryBytes) + // Both nearest slots are occupied by vectors whose join has no parent. + expect(nearestParents.all(queryBytes, 2)).toEqual([]) + return { ...store, parents, vectors, nearestParents, expectedHits } + }) + embedder.embedText.mockClear() + embedder.embedBatch.mockClear() + const infoSpy = vi.spyOn(logger, "info").mockImplementation(() => {}) + onTestFinished(() => infoSpy.mockRestore()) + + const rebuilt = await index.rebuildFromVault({ vaultPath }, logger) + await rebuilt.embedding + for (const store of retainedStores) { + expect(inspect.prepare(`SELECT * FROM ${store.parentTable} ORDER BY id`).all()).toEqual( + store.parents, + ) + expect( + inspect + .prepare( + `SELECT ${store.vectorKey}, hex(embedding) AS embedding FROM ${store.vectorTable} ORDER BY ${store.vectorKey}`, + ) + .all(), + ).toEqual(store.vectors) + expect(store.nearestParents.all(queryBytes, 2)).toEqual(store.expectedHits) + } + expect(infoSpy).toHaveBeenCalledWith("rebuilt index", { + count: 2, + totalBytes: 71, + orphanNoteVectorsRemoved: 2, + orphanFileVectorsRemoved: 2, + orphanMemoryVectorsRemoved: 2, + }) + expect(embedder.embedText).not.toHaveBeenCalled() + expect(embedder.embedBatch).not.toHaveBeenCalled() + + infoSpy.mockClear() + const repeated = await index.rebuildFromVault({ vaultPath }, logger) + await repeated.embedding + expect(infoSpy).toHaveBeenCalledWith("rebuilt index", { + count: 2, + totalBytes: 71, + orphanNoteVectorsRemoved: 0, + orphanFileVectorsRemoved: 0, + orphanMemoryVectorsRemoved: 0, + }) + for (const store of retainedStores) { + expect( + inspect + .prepare( + `SELECT ${store.vectorKey}, hex(embedding) AS embedding FROM ${store.vectorTable} ORDER BY ${store.vectorKey}`, + ) + .all(), + ).toEqual(store.vectors) + } + expect(embedder.embedText).not.toHaveBeenCalled() + expect(embedder.embedBatch).not.toHaveBeenCalled() + }) +}) + describe("file content vector embeddings", () => { const DIMENSIONS = 384 const createMockEmbedder = () => ({ @@ -6194,7 +6776,7 @@ describe("file content vector embeddings", () => { }) index.upsertNonMdFile("docs/overview.txt", 100) - index.upsertFileContent( + const sourceVersion = index.upsertFileContent( { filePath: "docs/overview.txt", rawContent: TEXT_FILE_CONTENT, @@ -6202,7 +6784,10 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/overview.txt" }, logger) + await index.embedFileContent( + { sourceVersion: sourceVersion, filePath: "docs/overview.txt" }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) }) @@ -6214,7 +6799,7 @@ describe("file content vector embeddings", () => { }) index.upsertNonMdFile("docs/overview.txt", 100) - index.upsertFileContent( + const sourceVersion = index.upsertFileContent( { filePath: "docs/overview.txt", rawContent: TEXT_FILE_CONTENT, @@ -6222,10 +6807,16 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/overview.txt" }, logger) + await index.embedFileContent( + { sourceVersion: sourceVersion, filePath: "docs/overview.txt" }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) - await index.embedFileContent({ filePath: "docs/overview.txt" }, logger) + await index.embedFileContent( + { sourceVersion: sourceVersion, filePath: "docs/overview.txt" }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) }) @@ -6236,7 +6827,7 @@ describe("file content vector embeddings", () => { }) index.upsertNonMdFile("docs/overview.txt", 100) - index.upsertFileContent( + const originalSourceVersion = index.upsertFileContent( { filePath: "docs/overview.txt", rawContent: TEXT_FILE_CONTENT, @@ -6244,10 +6835,13 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/overview.txt" }, logger) + await index.embedFileContent( + { sourceVersion: originalSourceVersion, filePath: "docs/overview.txt" }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) - index.upsertFileContent( + const updatedSourceVersion = index.upsertFileContent( { filePath: "docs/overview.txt", rawContent: "Completely different content about networking protocols.", @@ -6255,19 +6849,20 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/overview.txt" }, logger) + await index.embedFileContent( + { sourceVersion: updatedSourceVersion, filePath: "docs/overview.txt" }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(2) }) - it("is a no-op when file is not in file_content table", async () => { - const mockEmbedder = createMockEmbedder() - const index = createSearchIndex(":memory:", mockEmbedder, undefined, { - fileToolsEnabled: true, - }) - - await index.embedFileContent({ filePath: "nonexistent.txt" }, logger) - - expect(mockEmbedder.embedText).not.toHaveBeenCalled() + it("is a no-op when the captured file source was removed", async () => { + const fixture = await createEmbeddingRaceIndex("file") + const sourceVersion = fixture.upsert(TEXT_FILE_CONTENT) + fixture.remove() + expect(fixture.inspect.prepare("SELECT path FROM file_content").all()).toEqual([]) + await fixture.embed(TEXT_FILE_CONTENT, sourceVersion) + expect(fixture.embedder.embedText).not.toHaveBeenCalled() }) it("is a no-op when no embedder is provided", async () => { @@ -6276,7 +6871,7 @@ describe("file content vector embeddings", () => { }) index.upsertNonMdFile("docs/overview.txt", 100) - index.upsertFileContent( + const sourceVersion = index.upsertFileContent( { filePath: "docs/overview.txt", rawContent: TEXT_FILE_CONTENT, @@ -6286,7 +6881,10 @@ describe("file content vector embeddings", () => { ) await expect( - index.embedFileContent({ filePath: "docs/overview.txt" }, logger), + index.embedFileContent( + { sourceVersion: sourceVersion, filePath: "docs/overview.txt" }, + logger, + ), ).resolves.toBeUndefined() }) }) @@ -6303,7 +6901,7 @@ describe("file content vector embeddings", () => { }) index.upsertNonMdFile("docs/overview.txt", 100) - index.upsertFileContent( + const sourceVersion = index.upsertFileContent( { filePath: "docs/overview.txt", rawContent: TEXT_FILE_CONTENT, @@ -6311,7 +6909,10 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/overview.txt" }, logger) + await index.embedFileContent( + { sourceVersion: sourceVersion, filePath: "docs/overview.txt" }, + logger, + ) expect(mockEmbedder.embedText).toHaveBeenCalledTimes(1) const inspectDb = new Database(dbPath, { readonly: true }) @@ -6364,7 +6965,7 @@ describe("file content vector embeddings", () => { }).join("\n\n") index.upsertNonMdFile("docs/long.txt", 2000) - index.upsertFileContent( + const originalSourceVersion = index.upsertFileContent( { filePath: "docs/long.txt", rawContent: longContent, @@ -6372,7 +6973,10 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/long.txt" }, logger) + await index.embedFileContent( + { sourceVersion: originalSourceVersion, filePath: "docs/long.txt" }, + logger, + ) // Verify multiple chunks were created via a read-only inspection connection const inspectDb = new Database(dbPath, { readonly: true }) @@ -6389,7 +6993,7 @@ describe("file content vector embeddings", () => { expect(chunkCountBefore.count).toBeGreaterThan(1) // Replace with short content — produces exactly 1 chunk - index.upsertFileContent( + const updatedSourceVersion = index.upsertFileContent( { filePath: "docs/long.txt", rawContent: "Short content.", @@ -6397,7 +7001,10 @@ describe("file content vector embeddings", () => { }, logger, ) - await index.embedFileContent({ filePath: "docs/long.txt" }, logger) + await index.embedFileContent( + { sourceVersion: updatedSourceVersion, filePath: "docs/long.txt" }, + logger, + ) const chunkCountAfter = countRow( inspectDb @@ -6541,7 +7148,7 @@ describe("TOC source-path forwarding at the embed call sites", () => { ) const doneContent = Array.from({ length: 300 }, (_, wordIndex) => `done${wordIndex}`).join(" ") const noteContent = `## Active\n${activeContent}\n\n## Done\n${doneContent}` - forwardingIndex.upsertNote( + const originalSourceVersion = forwardingIndex.upsertNote( { filePath: "Folder Alpha/Sub/TASKS.md", rawContent: noteContent, @@ -6550,7 +7157,11 @@ describe("TOC source-path forwarding at the embed call sites", () => { logger, ) await forwardingIndex.embedNote( - { notePath: "Folder Alpha/Sub/TASKS.md", rawContent: noteContent }, + { + sourceVersion: originalSourceVersion, + notePath: "Folder Alpha/Sub/TASKS.md", + rawContent: noteContent, + }, logger, ) @@ -6562,7 +7173,7 @@ describe("TOC source-path forwarding at the embed call sites", () => { " ", ) forwardingIndex.upsertNonMdFile("Folder Alpha/data.csv", 100) - forwardingIndex.upsertFileContent( + const updatedSourceVersion = forwardingIndex.upsertFileContent( { filePath: "Folder Alpha/data.csv", rawContent: `## Metrics\n${metricsContent}\n\n## Notes\n${notesContent}`, @@ -6570,7 +7181,10 @@ describe("TOC source-path forwarding at the embed call sites", () => { }, logger, ) - await forwardingIndex.embedFileContent({ filePath: "Folder Alpha/data.csv" }, logger) + await forwardingIndex.embedFileContent( + { sourceVersion: updatedSourceVersion, filePath: "Folder Alpha/data.csv" }, + logger, + ) const inspect = new Database(dbPath, { readonly: true }) onTestFinished(() => { diff --git a/src/vault-mcp/search/file-watcher.ts b/src/vault-mcp/search/file-watcher.ts index 49f561f01..c9caf9a1c 100644 --- a/src/vault-mcp/search/file-watcher.ts +++ b/src/vault-mcp/search/file-watcher.ts @@ -51,129 +51,148 @@ export const startFileWatcher = ( search: SearchIndex, options?: FileWatcherOptions, ): Promise => { - // Serializes embedding per note path so overlapping chokidar events for the - // same file can't interleave and overwrite vectors with stale content. + // Serializing per path limits model concurrency; source versions reject + // obsolete results independently of the order jobs finish. const pendingEmbeds = new Map>() + const currentEvents = new Map() /** Indexes an added or modified file: non-md files land in the asset table; * notes are read from disk, upserted into the FTS index, and re-embedded * (embeds serialized per path via pendingEmbeds). */ const handleChange = async (filePath: string): Promise => { const relativePath = relative(vaultPath, filePath) + const eventToken = Symbol() + currentEvents.set(relativePath, eventToken) - if (!filePath.endsWith(".md")) { - const fileStat = await statOrNull(filePath) - - // Vanished between the watcher event and the stat — the unlink event - // that follows will remove any existing row. - if (!fileStat) return - search.upsertNonMdFile(relativePath, fileStat.size) - - // Canvas files are always read — link extraction is unconditional. - // PDF and text files are only read when file content FTS is enabled. - const extension = extname(filePath) - const isCanvas = extension === ".canvas" - const isIndexableNonCanvas = - search.fileContentIndexingEnabled && INDEXABLE_TEXT_EXTENSIONS.has(extension) - - if (isCanvas || isIndexableNonCanvas) { - try { - let contentToIndex: string - - if (extension === ".pdf") { - const buffer = await readFile(filePath) - const pdfData = new Uint8Array(buffer.buffer, buffer.byteOffset, buffer.byteLength) - const pdfResult = await extractPdfText(pdfData) - contentToIndex = pdfResult.text - } else { - contentToIndex = await readFile(filePath, "utf8") - } - search.upsertFileContent( - { - filePath: relativePath, - rawContent: contentToIndex, - fileStat: { mtimeMs: fileStat.mtimeMs, size: fileStat.size }, - }, - logger, - ) - - // Embed file content vectors — serialized per path via the same - // pendingEmbeds map (note paths end in .md, file paths don't). - // Reads the processed content from the file_content table. - const previousEmbed = pendingEmbeds.get(relativePath) ?? Promise.resolve() - const currentEmbed = previousEmbed - .catch((previousError) => { - logger.debug("previous file embed failed, proceeding with current", { - path: relativePath, - error: describeError(previousError), + try { + if (!filePath.endsWith(".md")) { + const fileStat = await statOrNull(filePath) + + if (currentEvents.get(relativePath) !== eventToken) return + // Vanished between the watcher event and the stat — the unlink event + // that follows will remove any existing row. + if (!fileStat) return + search.upsertNonMdFile(relativePath, fileStat.size) + + // Canvas files are always read — link extraction is unconditional. + // PDF and text files are only read when file content FTS is enabled. + const extension = extname(filePath) + const isCanvas = extension === ".canvas" + const isIndexableNonCanvas = + search.fileContentIndexingEnabled && INDEXABLE_TEXT_EXTENSIONS.has(extension) + + if (isCanvas || isIndexableNonCanvas) { + try { + let contentToIndex: string + + if (extension === ".pdf") { + const buffer = await readFile(filePath) + const pdfData = new Uint8Array(buffer.buffer, buffer.byteOffset, buffer.byteLength) + const pdfResult = await extractPdfText(pdfData) + contentToIndex = pdfResult.text + } else { + contentToIndex = await readFile(filePath, "utf8") + } + + if (currentEvents.get(relativePath) !== eventToken) return + + const sourceVersion = search.upsertFileContent( + { + filePath: relativePath, + rawContent: contentToIndex, + fileStat: { mtimeMs: fileStat.mtimeMs, size: fileStat.size }, + }, + logger, + ) + + // Embed file content vectors — serialized per path via the same + // pendingEmbeds map (note paths end in .md, file paths don't). + // Reads the processed content from the file_content table. + const previousEmbed = pendingEmbeds.get(relativePath) ?? Promise.resolve() + const currentEmbed = previousEmbed + .catch((previousError) => { + logger.debug("previous file embed failed, proceeding with current", { + path: relativePath, + error: describeError(previousError), + }) }) - }) - .then(() => search.embedFileContent({ filePath: relativePath }, logger)) - pendingEmbeds.set(relativePath, currentEmbed) - currentEmbed - .catch((embedError) => { - logger.warn("file content embedding failed", { - path: relativePath, - error: describeError(embedError), + .then(() => { + return search.embedFileContent({ filePath: relativePath, sourceVersion }, logger) }) + pendingEmbeds.set(relativePath, currentEmbed) + currentEmbed + .catch((embedError) => { + logger.warn("file content embedding failed", { + path: relativePath, + error: describeError(embedError), + }) + }) + .finally(() => { + if (pendingEmbeds.get(relativePath) === currentEmbed) { + pendingEmbeds.delete(relativePath) + } + }) + } catch (error) { + logger.warn("file content indexing failed", { + path: relativePath, + error: describeError(error), }) - .finally(() => { - if (pendingEmbeds.get(relativePath) === currentEmbed) { - pendingEmbeds.delete(relativePath) - } - }) - } catch (error) { - logger.warn("file content indexing failed", { - path: relativePath, - error: describeError(error), - }) + } } - } - logger.debug("indexed non-md file", { path: relativePath }) - return - } + logger.debug("indexed non-md file", { path: relativePath }) + return + } - try { - const [content, fileStat] = await Promise.all([readFile(filePath, "utf8"), stat(filePath)]) - search.upsertNote( - { - filePath: relativePath, - rawContent: content, - fileStat: { mtimeMs: fileStat.mtimeMs, size: fileStat.size }, - }, - logger, - ) - // Promise chain serializes embedding per path — if two events arrive for - // the same note, the second waits for the first to finish. .catch() - // swallows the previous rejection so a transient failure can't cascade - // and block subsequent embeds for this path. .finally() clears the map - // entry on success OR failure so a rejected promise can't permanently - // block that note from re-embedding. - const previousEmbed = pendingEmbeds.get(relativePath) ?? Promise.resolve() - const currentEmbed = previousEmbed - .catch((previousError) => { - logger.debug("previous embed failed, proceeding with current", { - path: relativePath, - error: describeError(previousError), + try { + const [content, fileStat] = await Promise.all([readFile(filePath, "utf8"), stat(filePath)]) + + if (currentEvents.get(relativePath) !== eventToken) return + + const sourceVersion = search.upsertNote( + { + filePath: relativePath, + rawContent: content, + fileStat: { mtimeMs: fileStat.mtimeMs, size: fileStat.size }, + }, + logger, + ) + // Each path queues model work; recovering a rejected predecessor + // allows later jobs to run after a transient failure. + const previousEmbed = pendingEmbeds.get(relativePath) ?? Promise.resolve() + const currentEmbed = previousEmbed + .catch((previousError) => { + logger.debug("previous embed failed, proceeding with current", { + path: relativePath, + error: describeError(previousError), + }) + }) + .then(() => { + return search.embedNote( + { notePath: relativePath, rawContent: content, sourceVersion }, + logger, + ) }) + pendingEmbeds.set(relativePath, currentEmbed) + // Awaiting the finally-derived promise routes its rejection to the + // catch; awaiting only currentEmbed would leave it unhandled. + await currentEmbed.finally(() => { + if (pendingEmbeds.get(relativePath) === currentEmbed) { + pendingEmbeds.delete(relativePath) + } }) - .then(() => search.embedNote({ notePath: relativePath, rawContent: content }, logger)) - pendingEmbeds.set(relativePath, currentEmbed) - // Await the .finally()-derived promise, not currentEmbed itself — - // .finally() returns a new promise that rejects with the same error, - // and awaiting it routes that rejection into the outer catch instead - // of leaving a second, unhandled rejection. - await currentEmbed.finally(() => { - if (pendingEmbeds.get(relativePath) === currentEmbed) { - pendingEmbeds.delete(relativePath) - } - }) - } catch (err) { - logger.error("failed to process file change", { - path: relativePath, - error: describeError(err), - }) + } catch (err) { + logger.error("failed to process file change", { + path: relativePath, + error: describeError(err), + }) + } + } finally { + // A newer handler may already have finished and removed its entry. + // Only the current event owns cleanup; absence still invalidates old reads. + if (currentEvents.get(relativePath) === eventToken) { + currentEvents.delete(relativePath) + } } } @@ -181,6 +200,7 @@ export const startFileWatcher = ( * files, the note tables (FTS, links, tasks, vectors) for notes. */ const handleDelete = (filePath: string): void => { const relativePath = relative(vaultPath, filePath) + currentEvents.delete(relativePath) if (!filePath.endsWith(".md")) { search.removeNonMdFile(relativePath) diff --git a/src/vault-mcp/search/search-index.ts b/src/vault-mcp/search/search-index.ts index fba19a431..2a3eb2d74 100644 --- a/src/vault-mcp/search/search-index.ts +++ b/src/vault-mcp/search/search-index.ts @@ -750,6 +750,20 @@ export const createSearchIndex = ( >("SELECT path, title, folder, mtime, bytes FROM file_content WHERE path = ?") : null + /** Symbols distinguish source lifecycles even when content and mtime repeat. */ + const sourceVersions = new Map() + const isCurrentSourceVersion = ( + params: { sourcePath: string; sourceVersion: symbol }, + logger: Logger, + ): boolean => { + const { sourcePath, sourceVersion } = params + const currentVersion = sourceVersions.get(sourcePath) + + if (currentVersion !== undefined && currentVersion === sourceVersion) return true + logger.debug("skipped obsolete embedding", { path: sourcePath }) + return false + } + // ── Vector prepared statements (conditional on embedder) ────── const upsertChunkStmt = embedder ? db.prepare( @@ -839,16 +853,6 @@ export const createSearchIndex = ( ) : null - // ── Rebuild staleness checks ───────────────────────────────────── - // Used by rebuildFromVault's embedding pass to skip notes/files the file - // watcher already re-indexed while the pass was running. - const selectNoteMtimeStmt = db.prepare<[string], { mtime: number }>( - "SELECT mtime FROM notes WHERE path = ?", - ) - const selectFileMtimeStmt = fileToolsEnabled - ? db.prepare<[string], { mtime: number }>("SELECT mtime FROM file_content WHERE path = ?") - : null - // ── Memory-entry prepared statements (conditional on memoryDir) ── const insertMemoryEntryStmt = memoryDir ? db.prepare( @@ -908,6 +912,21 @@ export const createSearchIndex = ( `DELETE FROM memory_entry_vectors WHERE entry_id IN (SELECT id FROM memory_entries WHERE file = ?)`, ) : null + const deleteOrphanNoteVectorsStmt = embedder + ? db.prepare("DELETE FROM note_vectors WHERE chunk_id NOT IN (SELECT id FROM note_chunks)") + : null + const deleteOrphanFileVectorsStmt = fileContentVectorEnabled + ? db.prepare( + "DELETE FROM file_content_vectors WHERE chunk_id NOT IN (SELECT id FROM file_content_chunks)", + ) + : null + const deleteOrphanMemoryVectorsStmt = + memoryDir && embedder + ? db.prepare( + "DELETE FROM memory_entry_vectors WHERE entry_id NOT IN (SELECT id FROM memory_entries)", + ) + : null + // Query side — memoryRecall's two retrieval legs, each returning whole // rows (the tie-break JOIN already reads memory_entries, so a separate // per-row hydration lookup would re-read the same data). @@ -1154,7 +1173,7 @@ export const createSearchIndex = ( fileStat: { mtimeMs: number; size: number } }, logger: Logger, - ): void => { + ): symbol => { const extension = posix.extname(params.filePath) const isCanvas = extension === ".canvas" @@ -1207,11 +1226,15 @@ export const createSearchIndex = ( } })() + const sourceVersion = Symbol() + sourceVersions.set(params.filePath, sourceVersion) + logger.debug("indexed file content", { path: params.filePath, links: canvasLinks.length, fts: Boolean(upsertFileContentStmt), }) + return sourceVersion } /** Removes a file's content from FTS and its links from the graph. */ @@ -1227,6 +1250,7 @@ export const createSearchIndex = ( } deleteLinksStmt.run(params.filePath) })() + sourceVersions.delete(params.filePath) logger.debug("removed file content", { path: params.filePath }) } @@ -1358,7 +1382,7 @@ export const createSearchIndex = ( skipLinks?: boolean }, logger: Logger, - ): void => { + ): symbol => { const { filePath, rawContent, fileStat } = params const skipLinks = params.skipLinks ?? false const parsed = parseNote(rawContent) @@ -1524,13 +1548,16 @@ export const createSearchIndex = ( } })() - // Emitted after the transaction commits so a rolled-back write can't - // leave a success line behind. + const sourceVersion = Symbol() + sourceVersions.set(filePath, sourceVersion) + + /** A failed upsert never publishes a version or a success log. */ logger.debug("indexed note", { path: note.path, bytes: note.bytes, tasksIndexed: extractedTasks.length, }) + return sourceVersion } // ── Embedding pipeline ───────────────────────────────────────── @@ -1539,7 +1566,7 @@ export const createSearchIndex = ( * gating skips chunks whose text hasn't changed since the last embedding. * Returns the number of chunks that were actually embedded (0 = all cached). */ const embedAndStoreChunks = async ( - params: { notePath: string; rawContent: string }, + params: { notePath: string; rawContent: string; sourceVersion: symbol }, logger: Logger, ): Promise => { const { notePath, rawContent } = params @@ -1556,6 +1583,11 @@ export const createSearchIndex = ( return 0 } + if ( + !isCurrentSourceVersion({ sourcePath: notePath, sourceVersion: params.sourceVersion }, logger) + ) + return 0 + const parsed = parseNote(rawContent) const noteTitle = (isString(parsed.data.title) ? parsed.data.title : null) ?? basename(notePath, ".md") @@ -1584,6 +1616,14 @@ export const createSearchIndex = ( let embeddedCount = 0 for (const chunk of chunks) { + if ( + !isCurrentSourceVersion( + { sourcePath: notePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return embeddedCount + const hash = contentHash(chunk.text) // Skip if content hasn't changed @@ -1591,6 +1631,14 @@ export const createSearchIndex = ( const embedding = await embedder.embedText(chunk.text) + if ( + !isCurrentSourceVersion( + { sourcePath: notePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return embeddedCount + // Wrap the DB writes in a transaction so the content hash is never // saved without its corresponding vector — prevents a crash between // chunk upsert and vector insert from permanently marking the chunk @@ -1624,6 +1672,11 @@ export const createSearchIndex = ( embeddedCount++ } + if ( + !isCurrentSourceVersion({ sourcePath: notePath, sourceVersion: params.sourceVersion }, logger) + ) + return embeddedCount + // Delete stale chunks and their vectors (note now has fewer chunks than before) if (deleteStaleVectorsStmt) { deleteStaleVectorsStmt.run(notePath, chunks.length) @@ -1652,23 +1705,40 @@ export const createSearchIndex = ( * file and section name ("Agents > Communication\n...") so both the * embedder and cross-encoder see which file an entry belongs to — the * date is excluded (semantic noise). Returns the number embedded. */ - const embedMemoryEntriesForFile = async (memoryFile: string, logger: Logger): Promise => { + const embedMemoryEntriesForFile = async ( + params: { memoryFile: string; notePath: string; sourceVersion: symbol }, + logger: Logger, + ): Promise => { + const { memoryFile, notePath, sourceVersion } = params + if (!embedder || !selectUnembeddedMemoryEntriesStmt || !insertMemoryVectorStmt) { return 0 } + if (!isCurrentSourceVersion({ sourcePath: notePath, sourceVersion }, logger)) return 0 + const unembeddedRows = selectUnembeddedMemoryEntriesStmt.all(memoryFile) if (unembeddedRows.length === 0) return 0 + /** Only completed batch writes count when a later batch becomes obsolete. */ + let embeddedCount = 0 + for ( let batchStart = 0; batchStart < unembeddedRows.length; batchStart += MEMORY_EMBED_BATCH_SIZE ) { + if (!isCurrentSourceVersion({ sourcePath: notePath, sourceVersion }, logger)) + return embeddedCount + const batchRows = unembeddedRows.slice(batchStart, batchStart + MEMORY_EMBED_BATCH_SIZE) const embeddings = await embedder.embedBatch( batchRows.map((row) => `${row.file} > ${row.section}\n${row.entry_text}`), ) + + if (!isCurrentSourceVersion({ sourcePath: notePath, sourceVersion }, logger)) + return embeddedCount + db.transaction(() => { for (const [rowIndexInBatch, row] of batchRows.entries()) { const embedding = embeddings[rowIndexInBatch] @@ -1685,13 +1755,14 @@ export const createSearchIndex = ( ) } })() + embeddedCount += batchRows.length } logger.debug("embedded memory entries", { file: memoryFile, - embeddedCount: unembeddedRows.length, + embeddedCount, }) - return unembeddedRows.length + return embeddedCount } /** Embed a note's content into vector storage — section-level chunks for @@ -1699,15 +1770,24 @@ export const createSearchIndex = ( * No-op when the embedding pipeline is disabled (no embedder provided). * Safe to call unconditionally. */ const embedNote = async ( - params: { notePath: string; rawContent: string }, + params: { notePath: string; rawContent: string; sourceVersion: symbol }, logger: Logger, ): Promise => { if (!embedder) return await embedAndStoreChunks(params, logger) const memoryFile = memoryFileNameFromPath(params.notePath) - if (memoryFile) { - await embedMemoryEntriesForFile(memoryFile, logger) + if ( + memoryFile && + isCurrentSourceVersion( + { sourcePath: params.notePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) { + await embedMemoryEntriesForFile( + { memoryFile, notePath: params.notePath, sourceVersion: params.sourceVersion }, + logger, + ) } } @@ -1715,7 +1795,7 @@ export const createSearchIndex = ( * Mirrors embedAndStoreChunks for notes: content-hash gated, transaction-wrapped, * stale chunks cleaned up. Returns the number of chunks actually (re-)embedded. */ const embedAndStoreFileChunks = async ( - params: { filePath: string; title: string; content: string }, + params: { filePath: string; title: string; content: string; sourceVersion: symbol }, logger: Logger, ): Promise => { if ( @@ -1730,6 +1810,14 @@ export const createSearchIndex = ( return 0 } + if ( + !isCurrentSourceVersion( + { sourcePath: params.filePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return 0 + // chunkContent handles file content too — sourcePath extracts folder // segments for the TOC chunk's disambiguation line. const chunks = chunkContent({ @@ -1747,12 +1835,28 @@ export const createSearchIndex = ( let embeddedCount = 0 for (const chunk of chunks) { + if ( + !isCurrentSourceVersion( + { sourcePath: params.filePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return embeddedCount + const hash = contentHash(chunk.text) if (existingHashes.get(chunk.index) === hash) continue const embedding = await embedder.embedText(chunk.text) + if ( + !isCurrentSourceVersion( + { sourcePath: params.filePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return embeddedCount + db.transaction(() => { const existingChunk = selectFileChunkIdStmt.get(params.filePath, chunk.index) @@ -1782,6 +1886,14 @@ export const createSearchIndex = ( embeddedCount++ } + if ( + !isCurrentSourceVersion( + { sourcePath: params.filePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return embeddedCount + if (deleteStaleFileVectorsStmt) { deleteStaleFileVectorsStmt.run(params.filePath, chunks.length) } @@ -1799,14 +1911,26 @@ export const createSearchIndex = ( * content from the file_content table (already linearized/truncated by * upsertFileContent). No-op when the embedding pipeline or file tools are * disabled, or the file is not in the FTS index. Safe to call unconditionally. */ - const embedFileContent = async (params: { filePath: string }, logger: Logger): Promise => { + const embedFileContent = async ( + params: { filePath: string; sourceVersion: symbol }, + logger: Logger, + ): Promise => { if (!embedder || !selectFileContentForEmbeddingStmt) return + if ( + !isCurrentSourceVersion( + { sourcePath: params.filePath, sourceVersion: params.sourceVersion }, + logger, + ) + ) + return + const fileContentRow = selectFileContentForEmbeddingStmt.get(params.filePath) if (!fileContentRow) return await embedAndStoreFileChunks( { filePath: params.filePath, + sourceVersion: params.sourceVersion, title: fileContentRow.title, content: fileContentRow.content, }, @@ -1836,6 +1960,7 @@ export const createSearchIndex = ( removeMemoryEntriesForFile(memoryFile) } })() + sourceVersions.delete(filePath) } /** @@ -1849,6 +1974,10 @@ export const createSearchIndex = ( logger: Logger, ): Promise<{ count: number; embedding: Promise }> => { const { vaultPath } = params + const orphanNoteVectorsRemoved = deleteOrphanNoteVectorsStmt?.run().changes ?? 0 + const orphanFileVectorsRemoved = deleteOrphanFileVectorsStmt?.run().changes ?? 0 + const orphanMemoryVectorsRemoved = deleteOrphanMemoryVectorsStmt?.run().changes ?? 0 + sourceVersions.clear() db.exec("DELETE FROM notes_fts") db.exec("DELETE FROM notes") db.exec("DELETE FROM links") @@ -2056,148 +2185,168 @@ export const createSearchIndex = ( // prevent the server from starting. const skippedNotePaths = new Set() + const notesForEmbedding: Array<{ + relativePath: string + content: string + sourceVersion: symbol + }> = [] + const fileVersionsForEmbedding = new Map() + // better-sqlite3: .transaction() returns a function; call it immediately - db.transaction(() => { - // Index non-markdown files so extensionless wikilinks to .canvas, .base, - // etc. are recognized as file references rather than broken note links. - const nonMdCount = indexNonMarkdownFiles(nonMarkdownFileSizes) - logger.debug("indexed non-md files", { count: nonMdCount }) - - // Pass 1: index all notes (content, frontmatter, FTS) — skip link - // extraction here; Pass 2 handles it with the complete path list. - for (const note of noteContents) { - try { - upsertNote( - { - filePath: note.relativePath, - rawContent: note.content, - fileStat: { mtimeMs: note.modifiedAtMs, size: note.sizeBytes }, - skipLinks: true, - }, - logger, - ) - } catch (error) { - skippedNotePaths.add(note.relativePath) - logger.warn("skipped malformed note during rebuild", { - path: note.relativePath, - error: describeError(error), - }) + try { + db.transaction(() => { + // Index non-markdown files so extensionless wikilinks to .canvas, .base, + // etc. are recognized as file references rather than broken note links. + const nonMdCount = indexNonMarkdownFiles(nonMarkdownFileSizes) + logger.debug("indexed non-md files", { count: nonMdCount }) + + // Pass 1: index all notes (content, frontmatter, FTS) — skip link + // extraction here; Pass 2 handles it with the complete path list. + for (const note of noteContents) { + try { + const sourceVersion = upsertNote( + { + filePath: note.relativePath, + rawContent: note.content, + fileStat: { mtimeMs: note.modifiedAtMs, size: note.sizeBytes }, + skipLinks: true, + }, + logger, + ) + notesForEmbedding.push({ + relativePath: note.relativePath, + content: note.content, + sourceVersion, + }) + } catch (error) { + skippedNotePaths.add(note.relativePath) + logger.warn("skipped malformed note during rebuild", { + path: note.relativePath, + error: describeError(error), + }) + } } - } - // Entry-index reconciliation for memory files deleted while the server - // was down: memory_entries is not wiped above (like the vector tables, - // its rows survive on content-hash identity), so files that vanished - // from disk leave orphaned entries the per-file upsert never touches. - if (selectDistinctMemoryFilesStmt) { - // Built from the disk listing (markdownFiles) on purpose: a note - // that failed to read or parse still exists on disk, and treating - // it as deleted here would permanently remove its memory_entries - // rows (the table survives rebuilds on content-hash identity). - const memoryFilesOnDisk = new Set( - markdownFiles - .map((file) => memoryFileNameFromPath(file.relativePath)) - .filter((fileName) => fileName !== null), - ) - const deletedMemoryFiles = selectDistinctMemoryFilesStmt - .all() - .map((row) => row.file) - .filter((fileName) => !memoryFilesOnDisk.has(fileName)) - for (const deletedFile of deletedMemoryFiles) { - removeMemoryEntriesForFile(deletedFile) - } - if (deletedMemoryFiles.length > 0) { - logger.info("cleaned up entries for deleted memory files", { - count: deletedMemoryFiles.length, - }) + // Entry-index reconciliation for memory files deleted while the server + // was down: memory_entries is not wiped above (like the vector tables, + // its rows survive on content-hash identity), so files that vanished + // from disk leave orphaned entries the per-file upsert never touches. + if (selectDistinctMemoryFilesStmt) { + // Built from the disk listing (markdownFiles) on purpose: a note + // that failed to read or parse still exists on disk, and treating + // it as deleted here would permanently remove its memory_entries + // rows (the table survives rebuilds on content-hash identity). + const memoryFilesOnDisk = new Set( + markdownFiles + .map((file) => memoryFileNameFromPath(file.relativePath)) + .filter((fileName) => fileName !== null), + ) + const deletedMemoryFiles = selectDistinctMemoryFilesStmt + .all() + .map((row) => row.file) + .filter((fileName) => !memoryFilesOnDisk.has(fileName)) + for (const deletedFile of deletedMemoryFiles) { + removeMemoryEntriesForFile(deletedFile) + } + if (deletedMemoryFiles.length > 0) { + logger.info("cleaned up entries for deleted memory files", { + count: deletedMemoryFiles.length, + }) + } } - } - // Pass 2: re-extract links now that all paths are in the notes table, - // resolving targets that the per-note upsertNote pass may have missed - // (e.g. Note A links to Note B, but Note B was indexed after Note A). - const allPaths = selectAllNotePathsStmt.all() - const pathList = allPaths.map((row) => row.path) + // Pass 2: re-extract links now that all paths are in the notes table, + // resolving targets that the per-note upsertNote pass may have missed + // (e.g. Note A links to Note B, but Note B was indexed after Note A). + const allPaths = selectAllNotePathsStmt.all() + const pathList = allPaths.map((row) => row.path) - db.exec("DELETE FROM links") - for (const note of noteContents) { - if (skippedNotePaths.has(note.relativePath)) continue - try { - const parsed = parseNote(note.content) - for (const rawTarget of links.extractAll(parsed.content, parsed.data)) { - const resolved = links.resolve({ - target: rawTarget, - allPaths: pathList, - sourcePath: note.relativePath, - }) + db.exec("DELETE FROM links") + for (const note of noteContents) { + if (skippedNotePaths.has(note.relativePath)) continue + try { + const parsed = parseNote(note.content) + for (const rawTarget of links.extractAll(parsed.content, parsed.data)) { + const resolved = links.resolve({ + target: rawTarget, + allPaths: pathList, + sourcePath: note.relativePath, + }) - if (resolved !== null) { - insertLinkStmt.run(note.relativePath, resolved) - } else { - const resolvedNonMdPath = resolveNonMarkdownFile(rawTarget, note.relativePath) - insertLinkStmt.run(note.relativePath, resolvedNonMdPath ?? rawTarget) + if (resolved !== null) { + insertLinkStmt.run(note.relativePath, resolved) + } else { + const resolvedNonMdPath = resolveNonMarkdownFile(rawTarget, note.relativePath) + insertLinkStmt.run(note.relativePath, resolvedNonMdPath ?? rawTarget) + } } + } catch (error) { + skippedNotePaths.add(note.relativePath) + logger.warn("skipped malformed note during rebuild", { + path: note.relativePath, + error: describeError(error), + }) } - } catch (error) { - skippedNotePaths.add(note.relativePath) - logger.warn("skipped malformed note during rebuild", { - path: note.relativePath, - error: describeError(error), - }) } - } - // File content indexing: canvas (FTS + link extraction), PDF and text - // (FTS only). upsertFileContent deletes old canvas links before - // inserting, so canvas links append cleanly after the note link pass. - const allFileContents = [...canvasContents, ...pdfContents, ...textFileContents] - for (const fileEntry of allFileContents) { - try { - upsertFileContent( - { - filePath: fileEntry.relativePath, - rawContent: fileEntry.content, - fileStat: { - mtimeMs: fileEntry.modifiedAtMs, - size: fileEntry.sizeBytes, + // File content indexing: canvas (FTS + link extraction), PDF and text + // (FTS only). upsertFileContent deletes old canvas links before + // inserting, so canvas links append cleanly after the note link pass. + const allFileContents = [...canvasContents, ...pdfContents, ...textFileContents] + for (const fileEntry of allFileContents) { + try { + const sourceVersion = upsertFileContent( + { + filePath: fileEntry.relativePath, + rawContent: fileEntry.content, + fileStat: { + mtimeMs: fileEntry.modifiedAtMs, + size: fileEntry.sizeBytes, + }, }, - }, - logger, - ) - } catch (error) { - logger.warn("skipped malformed file during rebuild", { - path: fileEntry.relativePath, - error: describeError(error), - }) + logger, + ) + fileVersionsForEmbedding.set(fileEntry.relativePath, sourceVersion) + } catch (error) { + logger.warn("skipped malformed file during rebuild", { + path: fileEntry.relativePath, + error: describeError(error), + }) + } } - } - })() + })() + } catch (error) { + /** Nested upserts can publish versions before the outer transaction commits. */ + sourceVersions.clear() + throw error + } const indexedNotes = noteContents.filter((note) => !skippedNotePaths.has(note.relativePath)) const totalBytes = indexedNotes.reduce((sum, note) => sum + note.sizeBytes, 0) logger.info("rebuilt index", { count: indexedNotes.length, totalBytes, + orphanNoteVectorsRemoved, + orphanFileVectorsRemoved, + orphanMemoryVectorsRemoved, ...(skippedNotePaths.size > 0 ? { skipped: skippedNotePaths.size } : {}), }) - // Extract only what Pass 3 needs so the full noteContents array (with - // every note's body + stats) can be garbage-collected during embedding. - const notesForEmbedding = noteContents.map((note) => ({ - relativePath: note.relativePath, - content: note.content, - snapshotMtimeMs: note.modifiedAtMs, - })) - // File content for embedding — read the processed text from file_content // (already linearized/extracted/truncated by upsertFileContent above). const filesForEmbedding = fileContentVectorEnabled ? db - .prepare( - "SELECT path, title, content, mtime FROM file_content", + .prepare( + "SELECT path, title, content FROM file_content", ) .all() + .flatMap((file) => { + const sourceVersion = fileVersionsForEmbedding.get(file.path) + + return sourceVersion === undefined ? [] : [{ ...file, sourceVersion }] + }) : [] + const readableNotePaths = new Set(noteContents.map((note) => note.relativePath)) // Pass 3 runs in the background — the server can start accepting requests // immediately after FTS indexing (Passes 1+2) finishes. Embedding is a @@ -2205,13 +2354,12 @@ export const createSearchIndex = ( const embeddingPromise = embedder ? (async () => { // Clean up vectors for notes that no longer exist on disk - const currentPaths = new Set(notesForEmbedding.map((note) => note.relativePath)) const indexedChunkPaths = db .prepare("SELECT DISTINCT note_path FROM note_chunks") .all() .map((row) => row.note_path) - const deletedPaths = indexedChunkPaths.filter((path) => !currentPaths.has(path)) + const deletedPaths = indexedChunkPaths.filter((path) => !readableNotePaths.has(path)) const hasDeletedNotes = deletedPaths.length > 0 && deleteVectorsForNoteStmt && deleteChunksForNoteStmt @@ -2230,24 +2378,36 @@ export const createSearchIndex = ( let entriesEmbedded = 0 let embedErrors = 0 for (const note of notesForEmbedding) { - // The watcher can update the index between the Pass 1 snapshot and - // this embed — skip the note if its mtime changed. - const currentNote = selectNoteMtimeStmt.get(note.relativePath) - const noteIsStale = !currentNote || currentNote.mtime !== note.snapshotMtimeMs - - if (noteIsStale) { + if ( + !isCurrentSourceVersion( + { sourcePath: note.relativePath, sourceVersion: note.sourceVersion }, + logger, + ) + ) continue - } try { chunksEmbedded += await embedAndStoreChunks( - { notePath: note.relativePath, rawContent: note.content }, + { + notePath: note.relativePath, + rawContent: note.content, + sourceVersion: note.sourceVersion, + }, logger, ) const memoryFile = memoryFileNameFromPath(note.relativePath) - if (memoryFile) { - entriesEmbedded += await embedMemoryEntriesForFile(memoryFile, logger) + if ( + memoryFile && + isCurrentSourceVersion( + { sourcePath: note.relativePath, sourceVersion: note.sourceVersion }, + logger, + ) + ) { + entriesEmbedded += await embedMemoryEntriesForFile( + { memoryFile, notePath: note.relativePath, sourceVersion: note.sourceVersion }, + logger, + ) } } catch (err) { embedErrors++ @@ -2301,20 +2461,23 @@ export const createSearchIndex = ( } } - if (filesForEmbedding.length > 0 && selectFileMtimeStmt) { + if (filesForEmbedding.length > 0) { let fileChunksEmbedded = 0 let fileEmbedErrors = 0 for (const file of filesForEmbedding) { - // Same watcher-race guard as for notes above. - const currentFile = selectFileMtimeStmt.get(file.path) - const fileIsStale = !currentFile || currentFile.mtime !== file.mtime - - if (fileIsStale) continue + if ( + !isCurrentSourceVersion( + { sourcePath: file.path, sourceVersion: file.sourceVersion }, + logger, + ) + ) + continue try { fileChunksEmbedded += await embedAndStoreFileChunks( { filePath: file.path, + sourceVersion: file.sourceVersion, title: file.title, content: file.content, }, From 302aaf234463ccd3db9066da6c0118f198a18cbb Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:27:41 -0400 Subject: [PATCH 02/13] fix(review): document embedding freshness and startup cleanup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-review · gpt-6.1-sol --- ARCHITECTURE.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index eb1cd375b..54680c7a3 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -468,10 +468,18 @@ When no embedder is configured (`EMBEDDING_ENABLED=false`), no vectors are index 2. **Pass 2** — extract links (with the complete path list for resolution), then index file content (canvas, PDF, text → FTS5) 3. **Pass 3 (background)** — embed notes, then file content. Search works with FTS-only until vectors are ready +**Startup cleanup:** before resetting the source tables, the rebuild removes vectors whose parent chunk or memory-entry row is missing and retains vectors with surviving parents. + Vector tables persist across restarts and rebuilds (only FTS, notes, links, tasks, non-md, and file content tables are cleared). Pass 3 cleans up vectors for deleted notes and files, then embeds only new or modified chunks via content-hash gating. **Incremental updates:** the file watcher calls `embedNote` after `upsertNote` and `embedFileContent` after `upsertFileContent`; deletion cleans up both vectors and chunks. +**Embedding freshness:** + +- Each successful source upsert returns a unique `sourceVersion`. Queued watcher jobs and background rebuild snapshots retain that version, so deletion or replacement invalidates earlier work even when content or modification time repeats. +- Note, file-content and memory-entry writers check the captured version before model work and after each model await. Obsolete jobs skip derived writes, note/file tail pruning and later memory batches. +- The watcher assigns an event token before reading each file and checks it before indexing. Unlink or a newer event invalidates earlier reads; embedding stays serialized per path to limit model concurrency. + **Embedding pipeline:** Controlled by `EMBEDDING_ENABLED` (default: `true`). Markdown syntax is stripped before embedding (`plaintext.ts`). Short notes (under 500 body tokens) stay a single title-prefixed chunk. Longer notes split into per-heading sections via `chunker.ts`: - **Two views per note:** each top-level heading spans its full subtree (the aggregate view, so child text embeds twice); deeper headings own only the lines above the next heading of any level (the disjoint leaf view) From 63975c61165985834cd9f90e49ccea3956524ea7 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:44:25 -0400 Subject: [PATCH 03/13] style: clarify embedding lifetimes and index conventions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: code-quality · gpt-6.1-sol --- ARCHITECTURE.md | 6 +- src/vault-mcp/search/file-watcher.ts | 27 +- src/vault-mcp/search/search-index.ts | 580 ++++++++++++++------------- 3 files changed, 324 insertions(+), 289 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 54680c7a3..f6cd239a1 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -117,7 +117,9 @@ graph TB **Hybrid query:** MCP client → `vault_search` → FTS5 BM25 ranks (notes + file content) + sqlite-vec KNN ranks (notes + file content) → RRF fusion → cross-encoder reranking → response. -**Invariant — vault is source of truth:** The vault `.md` files are canonical. SQLite FTS5 is derived — rebuildable from scratch. Never write to the index directly. The sqlite-vec embeddings are equally derived — they persist across rebuilds as an optimization but can always be regenerated from the vault. +**Invariant — vault is source of truth:** Vault files are canonical. MCP content edits write to those files, and the watcher and startup rebuild derive the SQLite search index from them. + +Embeddings persist across rebuilds to reuse unchanged content and can be regenerated from the vault. ## MCP Tools @@ -484,7 +486,7 @@ Vector tables persist across restarts and rebuilds (only FTS, notes, links, task - **Two views per note:** each top-level heading spans its full subtree (the aggregate view, so child text embeds twice); deeper headings own only the lines above the next heading of any level (the disjoint leaf view) - **Chunk prefixes:** every fragment starts with the note title; aggregate and leaf fragments add a `Section:` line naming the heading's ancestor path (capped at the remaining token budget — leading ancestors are dropped when deep nesting with long names would floor the body budget, keeping the deepest segments; the Section line is suppressed entirely when the title and metadata exhaust the budget), while preamble fragments, a singleton wrapper's aggregate, and the TOC chunk keep the bare title. A heading whose slice is empty emits no section chunk — its name still rides the TOC chunk, and descendant chunks' Section lines carry it when it has children. A top-level heading with children but no body of its own still emits its aggregate (the slice spans the subtree) -- **Table-of-contents chunk:** each split note with named headings emits one short chunk (folder segments + title on one line, then heading names in document order, truncated at the chunk budget). Generic intent-phrased queries are structurally won by short chunks under the embedding model, so every split note gets one deliberately short chunk, made unique by its folder path +- **Table-of-contents chunk:** each split note with named headings emits one short chunk with its folder path and title on the first line, followed by heading names in document order within the chunk budget. This gives a broad query a compact view of the note's topics without requiring one section to represent the whole note - **Sub-splitting:** oversized sections split at paragraph boundaries (MAX_CHUNK_TOKENS = 450, minus each chunk's prefix cost), with a sub-minimum trailing fragment merged backward Content-hash gating (SHA-256 per chunk) skips re-embedding unchanged content on both incremental file-watcher updates and full rebuilds. diff --git a/src/vault-mcp/search/file-watcher.ts b/src/vault-mcp/search/file-watcher.ts index c9caf9a1c..10b9bb49c 100644 --- a/src/vault-mcp/search/file-watcher.ts +++ b/src/vault-mcp/search/file-watcher.ts @@ -54,6 +54,9 @@ export const startFileWatcher = ( // Serializing per path limits model concurrency; source versions reject // obsolete results independently of the order jobs finish. const pendingEmbeds = new Map>() + + /** Event tokens reject superseded filesystem reads until their handler finishes. + * Source versions belong to the index and protect model work after the handler returns. */ const currentEvents = new Map() /** Indexes an added or modified file: non-md files land in the asset table; @@ -83,16 +86,15 @@ export const startFileWatcher = ( if (isCanvas || isIndexableNonCanvas) { try { - let contentToIndex: string + const readContentToIndex = async (): Promise => { + if (extension !== ".pdf") return await readFile(filePath, "utf8") - if (extension === ".pdf") { const buffer = await readFile(filePath) const pdfData = new Uint8Array(buffer.buffer, buffer.byteOffset, buffer.byteLength) const pdfResult = await extractPdfText(pdfData) - contentToIndex = pdfResult.text - } else { - contentToIndex = await readFile(filePath, "utf8") + return pdfResult.text } + const contentToIndex = await readContentToIndex() if (currentEvents.get(relativePath) !== eventToken) return @@ -120,6 +122,9 @@ export const startFileWatcher = ( return search.embedFileContent({ filePath: relativePath, sourceVersion }, logger) }) pendingEmbeds.set(relativePath, currentEmbed) + + /** This detached job owns its error reporting and queue cleanup after the + * file handler returns; an older job must leave a newer queued job registered. */ currentEmbed .catch((embedError) => { logger.warn("file content embedding failed", { @@ -174,8 +179,9 @@ export const startFileWatcher = ( ) }) pendingEmbeds.set(relativePath, currentEmbed) - // Awaiting the finally-derived promise routes its rejection to the - // catch; awaiting only currentEmbed would leave it unhandled. + + /** This awaited job reports failures through the handler's catch. Await the + * cleanup promise too, so its rejection is handled and newer queued jobs remain registered. */ await currentEmbed.finally(() => { if (pendingEmbeds.get(relativePath) === currentEmbed) { pendingEmbeds.delete(relativePath) @@ -309,9 +315,10 @@ export const startFileWatcher = ( dirPath: string, visitedRealPaths: Set, ): Promise => { + /** The wrapper lists descendants recursively; symlinked directories are followed below. */ const entries = await readdirOrNull(dirPath) - if (entries === null) { + if (!entries) { logger.debug("rescan skipped, directory vanished", { path: relative(vaultPath, dirPath), }) @@ -325,7 +332,7 @@ export const startFileWatcher = ( // — the same benign race as the vanished-listing branch, not an error. const realDirPath = await realpathOrNull(dirPath) - if (realDirPath === null) { + if (!realDirPath) { logger.debug("rescan skipped, directory vanished", { path: relative(vaultPath, dirPath), }) @@ -386,7 +393,7 @@ export const startFileWatcher = ( // directory), registration emits no replay for the write — retry once // it settles instead. if (isWithinStabilityWindow(fileStat.mtimeMs)) { - const parentWatchedBeforeRescan = trackedSiblings !== undefined + const parentWatchedBeforeRescan = Boolean(trackedSiblings) if (parentWatchedBeforeRescan) continue scheduleUnstableFileRetry(fullPath) diff --git a/src/vault-mcp/search/search-index.ts b/src/vault-mcp/search/search-index.ts index 2a3eb2d74..375dff1b7 100644 --- a/src/vault-mcp/search/search-index.ts +++ b/src/vault-mcp/search/search-index.ts @@ -335,10 +335,8 @@ export const createSearchIndex = ( embedder?: Embedder, reranker?: Reranker, options?: { - /** Vault-relative memory folder ("About Me"). When set, dated entries in - * its direct-child .md files are additionally indexed at entry - * granularity for vault_memory_recall; undefined (memory disabled) - * skips the entry tables entirely. */ + /** A nonempty vault-relative folder enables direct-child memory entry indexing. + * Undefined skips the tables; an empty string creates tables with no entry operations. */ memoryDir?: string | undefined /** When true, creates file_content + file_content_fts tables for * full-text search of non-markdown file content (e.g. canvas). */ @@ -750,7 +748,10 @@ export const createSearchIndex = ( >("SELECT path, title, folder, mtime, bytes FROM file_content WHERE path = ?") : null - /** Symbols distinguish source lifecycles even when content and mtime repeat. */ + /** Pending embeddings retain their source write's symbol because content and mtime can repeat. + * - upsertNote and upsertFileContent replace the path's symbol after successful writes. + * - removeNote and removeFileContent remove it after successful deletes. + * - rebuildFromVault clears every symbol before rebuilding and on transaction rollback. */ const sourceVersions = new Map() const isCurrentSourceVersion = ( params: { sourcePath: string; sourceVersion: symbol }, @@ -759,7 +760,8 @@ export const createSearchIndex = ( const { sourcePath, sourceVersion } = params const currentVersion = sourceVersions.get(sourcePath) - if (currentVersion !== undefined && currentVersion === sourceVersion) return true + if (currentVersion === sourceVersion) return true + logger.debug("skipped obsolete embedding", { path: sourcePath }) return false } @@ -927,6 +929,21 @@ export const createSearchIndex = ( ) : null + const selectAllNoteChunkPathsStmt = embedder + ? db.prepare<[], { note_path: string }>("SELECT DISTINCT note_path FROM note_chunks") + : null + const selectAllFileContentForEmbeddingStmt = fileContentVectorEnabled + ? db.prepare<[], { path: string; title: string; content: string }>( + "SELECT path, title, content FROM file_content", + ) + : null + const selectAllFileContentPathsStmt = fileContentVectorEnabled + ? db.prepare<[], { path: string }>("SELECT path FROM file_content") + : null + const selectAllFileChunkPathsStmt = fileContentVectorEnabled + ? db.prepare<[], { file_path: string }>("SELECT DISTINCT file_path FROM file_content_chunks") + : null + // Query side — memoryRecall's two retrieval legs, each returning whole // rows (the tie-break JOIN already reads memory_entries, so a separate // per-row hydration lookup would re-read the same data). @@ -1058,7 +1075,11 @@ export const createSearchIndex = ( * full-filename tiers hit an extensionless file (LICENSE, Dockerfile) by * exact path or path suffix, and the stem tiers hit any file whose * extension-stripped name matches. */ - const resolveNonMarkdownFile = (target: string, sourcePath?: string): string | null => { + const resolveNonMarkdownFile = (params: { + target: string + sourcePath?: string + }): string | null => { + const { target, sourcePath } = params const relativeTarget = sourcePath === undefined ? null : posix.join(posix.dirname(sourcePath), target) @@ -1138,7 +1159,7 @@ export const createSearchIndex = ( // raw targets (e.g. "Trip Route") to resolved paths ("Trip Route.canvas"). const unresolvedLinks = selectUnresolvedLinksStmt.all() for (const link of unresolvedLinks) { - const resolvedPath = resolveNonMarkdownFile(link.target, link.source) + const resolvedPath = resolveNonMarkdownFile({ target: link.target, sourcePath: link.source }) if (resolvedPath !== null) { updateLinkTargetStmt.run({ @@ -1160,12 +1181,9 @@ export const createSearchIndex = ( // ── File content indexing ───────────────────────────────────── - /** Indexes non-markdown file content for FTS and extracts graph links. - * Canvas files pass raw JSON — linearized here for FTS and parsed for - * link extraction. Other file types pass pre-rendered text (PDF text, - * raw UTF-8) — inserted into FTS as-is, no link extraction. FTS - * indexing is gated behind `fileToolsEnabled`; canvas link extraction - * is unconditional (graph integrity is a core feature). */ + /** Pass the returned source version to embedFileContent for this upsert. + * - Canvas JSON is linearized for FTS and parsed for graph links regardless of fileToolsEnabled. + * - Other files supply rendered text; FTS storage requires fileToolsEnabled and caps UTF-8 bytes. */ const upsertFileContent = ( params: { filePath: string @@ -1237,7 +1255,7 @@ export const createSearchIndex = ( return sourceVersion } - /** Removes a file's content from FTS and its links from the graph. */ + /** Removes a file's FTS content, graph links, chunks and vectors, invalidating pending embeds. */ const removeFileContent = (params: { filePath: string }, logger: Logger): void => { db.transaction(() => { if (deleteFileContentStmt && deleteFileContentFtsStmt) { @@ -1284,7 +1302,12 @@ export const createSearchIndex = ( * (delete-then-insert, the notes_fts convention). Runs in one transaction; * embedding is NOT gated on these hashes but on vector absence, so a crash * between this upsert and embedMemoryEntriesForFile self-heals. */ - const upsertMemoryEntries = (memoryFile: string, noteBody: string, logger: Logger): void => { + const upsertMemoryEntries = ( + params: { memoryFile: string; noteBody: string }, + logger: Logger, + ): void => { + const { memoryFile, noteBody } = params + if ( !insertMemoryEntryStmt || !updateMemoryEntryIndexStmt || @@ -1373,7 +1396,7 @@ export const createSearchIndex = ( // FTS rows are managed manually (delete-then-insert) because SQLite triggers // combined with INSERT OR REPLACE cause FTS5 corruption. - /** Parses a note's content and frontmatter, then indexes it for search. */ + /** Indexes a note and returns the source version to pass with the same content to embedNote. */ const upsertNote = ( params: { filePath: string @@ -1399,14 +1422,14 @@ export const createSearchIndex = ( // Detect Kanban done lanes for boards with kanban-plugin frontmatter. // The Kanban plugin marks completion lanes with a **Complete** paragraph. - const isKanbanBoard = Boolean(frontmatter["kanban-plugin"]) - let kanbanDoneLanes: string | null = null + const getKanbanDoneLanes = (): string | null => { + if (!frontmatter["kanban-plugin"]) return null - if (isKanbanBoard) { const headings = parseHeadings(bodyLines) const doneLanes = tasks.extractDoneLanes(bodyLines, headings) - kanbanDoneLanes = doneLanes.length > 0 ? JSON.stringify(doneLanes) : null + return doneLanes.length > 0 ? JSON.stringify(doneLanes) : null } + const kanbanDoneLanes = getKanbanDoneLanes() const note = { path: filePath, @@ -1498,7 +1521,7 @@ export const createSearchIndex = ( // Memory files additionally maintain their entry-granular index. Placed // before the skipLinks return so rebuild Pass 1 covers it. if (memoryFile) { - upsertMemoryEntries(memoryFile, parsed.content, logger) + upsertMemoryEntries({ memoryFile, noteBody: parsed.content }, logger) } if (skipLinks) return @@ -1516,10 +1539,14 @@ export const createSearchIndex = ( if (resolved !== null) { insertLinkStmt.run(note.path, resolved) - } else { - const resolvedNonMdPath = resolveNonMarkdownFile(rawTarget, note.path) - insertLinkStmt.run(note.path, resolvedNonMdPath ?? rawTarget) + continue } + + const resolvedNonMdPath = resolveNonMarkdownFile({ + target: rawTarget, + sourcePath: note.path, + }) + insertLinkStmt.run(note.path, resolvedNonMdPath ?? rawTarget) } // Re-resolve links still stored as raw text now that this note exists. @@ -1551,7 +1578,6 @@ export const createSearchIndex = ( const sourceVersion = Symbol() sourceVersions.set(filePath, sourceVersion) - /** A failed upsert never publishes a version or a success log. */ logger.debug("indexed note", { path: note.path, bytes: note.bytes, @@ -1562,9 +1588,8 @@ export const createSearchIndex = ( // ── Embedding pipeline ───────────────────────────────────────── - /** Chunk, hash, embed, and store vectors for a single note. Content-hash - * gating skips chunks whose text hasn't changed since the last embedding. - * Returns the number of chunks that were actually embedded (0 = all cached). */ + /** The return counts changed chunks committed before completion or obsolescence. + * Zero covers disabled, cached and obsolete calls. */ const embedAndStoreChunks = async ( params: { notePath: string; rawContent: string; sourceVersion: symbol }, logger: Logger, @@ -1677,7 +1702,8 @@ export const createSearchIndex = ( ) return embeddedCount - // Delete stale chunks and their vectors (note now has fewer chunks than before) + /** chunkContent assigns indices 0 through length - 1, so a shortened note's + * old tail starts at chunks.length; remove its vectors before their parent rows. */ if (deleteStaleVectorsStmt) { deleteStaleVectorsStmt.run(notePath, chunks.length) } @@ -1697,14 +1723,10 @@ export const createSearchIndex = ( * pipeline call per entry to one per 16. */ const MEMORY_EMBED_BATCH_SIZE = 16 - /** Embeds every not-yet-embedded entry of one memory file. Table-driven: - * upsertMemoryEntries is the single parse of truth, and this reads entry - * texts straight from memory_entries WHERE no vector exists — gating on - * vector ABSENCE rather than content hashes, so a crash between upsert and - * embed self-heals on the next call. The embedding input prefixes the - * file and section name ("Agents > Communication\n...") so both the - * embedder and cross-encoder see which file an entry belongs to — the - * date is excluded (semantic noise). Returns the number embedded. */ + /** Missing vectors are retried on the next call if embedding stops after source indexing. + * - memoryFile is the bare name ("Agents"); notePath is its full vault-relative .md path. + * - Model input includes the file and section for context and excludes the date. + * - The return counts committed entries, including completed batches before obsolescence. */ const embedMemoryEntriesForFile = async ( params: { memoryFile: string; notePath: string; sourceVersion: symbol }, logger: Logger, @@ -1714,6 +1736,7 @@ export const createSearchIndex = ( if (!embedder || !selectUnembeddedMemoryEntriesStmt || !insertMemoryVectorStmt) { return 0 } + if (!isCurrentSourceVersion({ sourcePath: notePath, sourceVersion }, logger)) return 0 const unembeddedRows = selectUnembeddedMemoryEntriesStmt.all(memoryFile) @@ -1791,9 +1814,8 @@ export const createSearchIndex = ( } } - /** Chunks and embeds file content into file_content_chunks / file_content_vectors. - * Mirrors embedAndStoreChunks for notes: content-hash gated, transaction-wrapped, - * stale chunks cleaned up. Returns the number of chunks actually (re-)embedded. */ + /** The return counts changed chunks committed before completion or obsolescence. + * Zero covers disabled, cached and obsolete calls. */ const embedAndStoreFileChunks = async ( params: { filePath: string; title: string; content: string; sourceVersion: symbol }, logger: Logger, @@ -1894,6 +1916,8 @@ export const createSearchIndex = ( ) return embeddedCount + /** chunkContent assigns indices 0 through length - 1, so a shortened file's + * old tail starts at chunks.length; remove its vectors before their parent rows. */ if (deleteStaleFileVectorsStmt) { deleteStaleFileVectorsStmt.run(params.filePath, chunks.length) } @@ -1963,12 +1987,8 @@ export const createSearchIndex = ( sourceVersions.delete(filePath) } - /** - * - Rebuilds note, task, link and file-content indexes from visible vault files. - * - Retains vectors and memory entries to reuse unchanged embeddings. - * - Returns the note count and a background embedding promise so requests - * can start before embedding finishes. - */ + /** Rebuilds note, task, file and graph indexes; reconciles retained memory rows and vectors. + * Returns the note count and a background embedding promise. */ const rebuildFromVault = async ( params: { vaultPath: string }, logger: Logger, @@ -1991,6 +2011,14 @@ export const createSearchIndex = ( // gating to skip unchanged chunks, so only new/modified notes re-embed. // Deleted notes are cleaned up in Pass 3 before embedding starts. + type RebuildFilePaths = { relativePath: string; absolutePath: string } + type RebuildFileContent = { + relativePath: string + content: string + modifiedAtMs: number + sizeBytes: number + } + const normalizedVault = resolve(vaultPath) const allEntries = await readdir(vaultPath, { recursive: true, @@ -2042,16 +2070,15 @@ export const createSearchIndex = ( // Stat non-md files before the write transaction (fs stays out of it). // A file vanishing between listing and stat (sync race) is dropped here // and re-indexed by its own watcher event. - const nonMarkdownFileSizes = ( - await Promise.all( - allNonMdFiles.map(async (file) => { - const fileStat = await statOrNull(file.absolutePath) - - if (!fileStat) return null - return { relativePath: file.relativePath, bytes: fileStat.size } - }), - ) - ).filter((entry) => entry !== null) + const readFileSize = async (file: RebuildFilePaths) => { + const fileStat = await statOrNull(file.absolutePath) + + if (!fileStat) return null + return { relativePath: file.relativePath, bytes: fileStat.size } + } + const nonMarkdownFileSizes = (await Promise.all(allNonMdFiles.map(readFileSize))).filter( + (entry) => entry !== null, + ) const canvasFiles = allNonMdFiles.filter((file) => file.relativePath.endsWith(".canvas")) // PDF and text files are only read when file content FTS is enabled — // without the tables, the extraction is wasted I/O. @@ -2066,30 +2093,31 @@ export const createSearchIndex = ( : [] // Read canvas files for content indexing + link extraction. - const canvasContents = ( - await Promise.all( - canvasFiles.map(async (file) => { - try { - const [content, fileStat] = await Promise.all([ - readFile(file.absolutePath, "utf8"), - stat(file.absolutePath), - ]) - return { - relativePath: file.relativePath, - content, - modifiedAtMs: fileStat.mtimeMs, - sizeBytes: fileStat.size, - } - } catch (error) { - logger.warn("skipped unreadable canvas file during rebuild", { - path: file.relativePath, - error: describeError(error), - }) - return null - } - }), - ) - ).filter((entry) => entry !== null) + const readCanvasContent = async ( + file: RebuildFilePaths, + ): Promise => { + try { + const [content, fileStat] = await Promise.all([ + readFile(file.absolutePath, "utf8"), + stat(file.absolutePath), + ]) + return { + relativePath: file.relativePath, + content, + modifiedAtMs: fileStat.mtimeMs, + sizeBytes: fileStat.size, + } + } catch (error) { + logger.warn("skipped unreadable canvas file during rebuild", { + path: file.relativePath, + error: describeError(error), + }) + return null + } + } + const canvasContents = (await Promise.all(canvasFiles.map(readCanvasContent))).filter( + (entry) => entry !== null, + ) // Extract PDF text with bounded concurrency (CPU-intensive pdfjs work). const extractPdfContent = async (file: { @@ -2130,55 +2158,55 @@ export const createSearchIndex = ( const pdfContents = pdfResults.filter((entry) => entry !== null) // Read text files for content indexing (raw UTF-8). - const textFileContents = ( - await Promise.all( - textFiles.map(async (file) => { - try { - const [content, fileStat] = await Promise.all([ - readFile(file.absolutePath, "utf8"), - stat(file.absolutePath), - ]) - return { - relativePath: file.relativePath, - content, - modifiedAtMs: fileStat.mtimeMs, - sizeBytes: fileStat.size, - } - } catch (error) { - logger.warn("skipped unreadable text file during rebuild", { - path: file.relativePath, - error: describeError(error), - }) - return null - } - }), - ) - ).filter((entry) => entry !== null) + const readTextFileContent = async ( + file: RebuildFilePaths, + ): Promise => { + try { + const [content, fileStat] = await Promise.all([ + readFile(file.absolutePath, "utf8"), + stat(file.absolutePath), + ]) + return { + relativePath: file.relativePath, + content, + modifiedAtMs: fileStat.mtimeMs, + sizeBytes: fileStat.size, + } + } catch (error) { + logger.warn("skipped unreadable text file during rebuild", { + path: file.relativePath, + error: describeError(error), + }) + return null + } + } + const textFileContents = (await Promise.all(textFiles.map(readTextFileContent))).filter( + (entry) => entry !== null, + ) - const noteContents = ( - await Promise.all( - markdownFiles.map(async (file) => { - try { - const [content, fileStat] = await Promise.all([ - readFile(file.absolutePath, "utf8"), - stat(file.absolutePath), - ]) - return { - relativePath: file.relativePath, - content, - modifiedAtMs: fileStat.mtimeMs, - sizeBytes: fileStat.size, - } - } catch (error) { - logger.warn("skipped unreadable note during rebuild", { - path: file.relativePath, - error: describeError(error), - }) - return null - } - }), - ) - ).filter((entry) => entry !== null) + const readNoteContent = async (file: RebuildFilePaths): Promise => { + try { + const [content, fileStat] = await Promise.all([ + readFile(file.absolutePath, "utf8"), + stat(file.absolutePath), + ]) + return { + relativePath: file.relativePath, + content, + modifiedAtMs: fileStat.mtimeMs, + sizeBytes: fileStat.size, + } + } catch (error) { + logger.warn("skipped unreadable note during rebuild", { + path: file.relativePath, + error: describeError(error), + }) + return null + } + } + const noteContents = (await Promise.all(markdownFiles.map(readNoteContent))).filter( + (entry) => entry !== null, + ) // Notes whose parse or index write throws are skipped with a warning // instead of aborting the rebuild — one malformed note must never @@ -2192,7 +2220,8 @@ export const createSearchIndex = ( }> = [] const fileVersionsForEmbedding = new Map() - // better-sqlite3: .transaction() returns a function; call it immediately + /** Nested upserts publish versions when their savepoints finish. If the outer + * transaction rolls back, clear those versions so no embed can use reverted source rows. */ try { db.transaction(() => { // Index non-markdown files so extensionless wikilinks to .canvas, .base, @@ -2200,8 +2229,8 @@ export const createSearchIndex = ( const nonMdCount = indexNonMarkdownFiles(nonMarkdownFileSizes) logger.debug("indexed non-md files", { count: nonMdCount }) - // Pass 1: index all notes (content, frontmatter, FTS) — skip link - // extraction here; Pass 2 handles it with the complete path list. + /** Pass 1 indexes note content before links can resolve against the complete path list. + * Queue every successful upsert for embedding, even if its later link pass fails. */ for (const note of noteContents) { try { const sourceVersion = upsertNote( @@ -2213,6 +2242,7 @@ export const createSearchIndex = ( }, logger, ) + notesForEmbedding.push({ relativePath: note.relativePath, content: note.content, @@ -2275,10 +2305,14 @@ export const createSearchIndex = ( if (resolved !== null) { insertLinkStmt.run(note.relativePath, resolved) - } else { - const resolvedNonMdPath = resolveNonMarkdownFile(rawTarget, note.relativePath) - insertLinkStmt.run(note.relativePath, resolvedNonMdPath ?? rawTarget) + continue } + + const resolvedNonMdPath = resolveNonMarkdownFile({ + target: rawTarget, + sourcePath: note.relativePath, + }) + insertLinkStmt.run(note.relativePath, resolvedNonMdPath ?? rawTarget) } } catch (error) { skippedNotePaths.add(note.relativePath) @@ -2316,11 +2350,11 @@ export const createSearchIndex = ( } })() } catch (error) { - /** Nested upserts can publish versions before the outer transaction commits. */ sourceVersions.clear() throw error } + /** The rebuild count excludes failures in either indexing pass; embedding uses Pass 1 successes. */ const indexedNotes = noteContents.filter((note) => !skippedNotePaths.has(note.relativePath)) const totalBytes = indexedNotes.reduce((sum, note) => sum + note.sizeBytes, 0) logger.info("rebuilt index", { @@ -2334,180 +2368,172 @@ export const createSearchIndex = ( // File content for embedding — read the processed text from file_content // (already linearized/extracted/truncated by upsertFileContent above). - const filesForEmbedding = fileContentVectorEnabled - ? db - .prepare( - "SELECT path, title, content FROM file_content", - ) - .all() - .flatMap((file) => { - const sourceVersion = fileVersionsForEmbedding.get(file.path) + const withFileSourceVersion = (file: { path: string; title: string; content: string }) => { + const sourceVersion = fileVersionsForEmbedding.get(file.path) - return sourceVersion === undefined ? [] : [{ ...file, sourceVersion }] - }) - : [] + return sourceVersion ? [{ ...file, sourceVersion }] : [] + } + const filesForEmbedding = + selectAllFileContentForEmbeddingStmt?.all().flatMap(withFileSourceVersion) ?? [] const readableNotePaths = new Set(noteContents.map((note) => note.relativePath)) // Pass 3 runs in the background — the server can start accepting requests // immediately after FTS indexing (Passes 1+2) finishes. Embedding is a // progressive enhancement: search works with FTS-only until vectors are ready. - const embeddingPromise = embedder - ? (async () => { - // Clean up vectors for notes that no longer exist on disk - const indexedChunkPaths = db - .prepare("SELECT DISTINCT note_path FROM note_chunks") - .all() - .map((row) => row.note_path) - - const deletedPaths = indexedChunkPaths.filter((path) => !readableNotePaths.has(path)) - const hasDeletedNotes = - deletedPaths.length > 0 && deleteVectorsForNoteStmt && deleteChunksForNoteStmt - - if (hasDeletedNotes) { - for (const path of deletedPaths) { - deleteVectorsForNoteStmt.run(path) - deleteChunksForNoteStmt.run(path) - } - logger.info("cleaned up vectors for deleted notes", { - count: deletedPaths.length, - }) - } - - // Running totals accumulated across the sequential embedding loop - let chunksEmbedded = 0 - let entriesEmbedded = 0 - let embedErrors = 0 - for (const note of notesForEmbedding) { - if ( - !isCurrentSourceVersion( - { sourcePath: note.relativePath, sourceVersion: note.sourceVersion }, - logger, - ) - ) - continue - - try { - chunksEmbedded += await embedAndStoreChunks( - { - notePath: note.relativePath, - rawContent: note.content, - sourceVersion: note.sourceVersion, - }, - logger, - ) - const memoryFile = memoryFileNameFromPath(note.relativePath) - - if ( - memoryFile && - isCurrentSourceVersion( - { sourcePath: note.relativePath, sourceVersion: note.sourceVersion }, - logger, - ) - ) { - entriesEmbedded += await embedMemoryEntriesForFile( - { memoryFile, notePath: note.relativePath, sourceVersion: note.sourceVersion }, - logger, - ) - } - } catch (err) { - embedErrors++ - logger.warn("failed to embed note", { - path: note.relativePath, - error: describeError(err), - }) - } - - // Yield to the event loop between notes so pending I/O callbacks - // (Express healthz, MCP tool handlers) can drain. - await setImmediateAsync() - } - // ── File content embedding ────────────────────────────── - // Clean up vectors for files that no longer exist — runs - // unconditionally so deletions while the server was down are caught - // even when filesForEmbedding is empty (mirroring the note-side cleanup). - if ( - fileContentVectorEnabled && - deleteFileVectorsForPathStmt && - deleteFileChunksForPathStmt - ) { - // Query the live table instead of the pre-loop snapshot — files - // the watcher indexed during the note embedding loop are in the - // table but absent from the snapshot. - const currentFilePaths = new Set( - db - .prepare("SELECT path FROM file_content") - .all() - .map((fileContentPathRow) => fileContentPathRow.path), - ) - const indexedFileChunkPaths = db - .prepare( - "SELECT DISTINCT file_path FROM file_content_chunks", - ) + const embeddingPromise = + embedder && selectAllNoteChunkPathsStmt + ? (async () => { + /** Keep chunks only for notes read into this rebuild's snapshot. + * Missing and unreadable notes lose their vectors; parse failures retain them. */ + const indexedChunkPaths = selectAllNoteChunkPathsStmt .all() - .map((chunkPathRow) => chunkPathRow.file_path) + .map((chunkPath) => chunkPath.note_path) - const deletedFilePaths = indexedFileChunkPaths.filter( - (path) => !currentFilePaths.has(path), - ) + const deletedPaths = indexedChunkPaths.filter((path) => !readableNotePaths.has(path)) + const hasDeletedNotes = + deletedPaths.length > 0 && deleteVectorsForNoteStmt && deleteChunksForNoteStmt - if (deletedFilePaths.length > 0) { - for (const path of deletedFilePaths) { - deleteFileVectorsForPathStmt.run(path) - deleteFileChunksForPathStmt.run(path) + if (hasDeletedNotes) { + for (const path of deletedPaths) { + deleteVectorsForNoteStmt.run(path) + deleteChunksForNoteStmt.run(path) } - logger.info("cleaned up vectors for deleted files", { - count: deletedFilePaths.length, + logger.info("cleaned up vectors for deleted notes", { + count: deletedPaths.length, }) } - } - if (filesForEmbedding.length > 0) { - let fileChunksEmbedded = 0 - let fileEmbedErrors = 0 - for (const file of filesForEmbedding) { + // Running totals accumulated across the sequential embedding loop + let chunksEmbedded = 0 + let entriesEmbedded = 0 + let embedErrors = 0 + for (const note of notesForEmbedding) { if ( !isCurrentSourceVersion( - { sourcePath: file.path, sourceVersion: file.sourceVersion }, + { sourcePath: note.relativePath, sourceVersion: note.sourceVersion }, logger, ) ) continue try { - fileChunksEmbedded += await embedAndStoreFileChunks( + chunksEmbedded += await embedAndStoreChunks( { - filePath: file.path, - sourceVersion: file.sourceVersion, - title: file.title, - content: file.content, + notePath: note.relativePath, + rawContent: note.content, + sourceVersion: note.sourceVersion, }, logger, ) + const memoryFile = memoryFileNameFromPath(note.relativePath) + + if ( + memoryFile && + isCurrentSourceVersion( + { sourcePath: note.relativePath, sourceVersion: note.sourceVersion }, + logger, + ) + ) { + entriesEmbedded += await embedMemoryEntriesForFile( + { memoryFile, notePath: note.relativePath, sourceVersion: note.sourceVersion }, + logger, + ) + } } catch (err) { - fileEmbedErrors++ - logger.warn("failed to embed file content", { - path: file.path, + embedErrors++ + logger.warn("failed to embed note", { + path: note.relativePath, error: describeError(err), }) } + // Yield to the event loop between notes so pending I/O callbacks + // (Express healthz, MCP tool handlers) can drain. await setImmediateAsync() } - logger.info("file content embedding pass complete", { - files: filesForEmbedding.length, - fileChunksEmbedded, - ...(fileEmbedErrors > 0 ? { fileEmbedErrors } : {}), - }) - } + // ── File content embedding ────────────────────────────── + // Clean up vectors for files that no longer exist — runs + // unconditionally so deletions while the server was down are caught + // even when filesForEmbedding is empty (mirroring the note-side cleanup). + const canCleanUpFileVectors = + fileContentVectorEnabled && + deleteFileVectorsForPathStmt && + deleteFileChunksForPathStmt && + selectAllFileContentPathsStmt && + selectAllFileChunkPathsStmt + + if (canCleanUpFileVectors) { + // Query the live table instead of the pre-loop snapshot — files + // the watcher indexed during the note embedding loop are in the + // table but absent from the snapshot. + const fileContentPaths = selectAllFileContentPathsStmt.all().map((file) => file.path) + const currentFilePaths = new Set(fileContentPaths) + const indexedFileChunkPaths = selectAllFileChunkPathsStmt + .all() + .map((chunkPath) => chunkPath.file_path) - logger.info("embedding pass complete", { - notes: notesForEmbedding.length, - chunksEmbedded, - ...(entriesEmbedded > 0 ? { entriesEmbedded } : {}), - ...(embedErrors > 0 ? { embedErrors } : {}), - }) - })() - : Promise.resolve() + const deletedFilePaths = indexedFileChunkPaths.filter( + (path) => !currentFilePaths.has(path), + ) + + if (deletedFilePaths.length > 0) { + for (const path of deletedFilePaths) { + deleteFileVectorsForPathStmt.run(path) + deleteFileChunksForPathStmt.run(path) + } + logger.info("cleaned up vectors for deleted files", { + count: deletedFilePaths.length, + }) + } + } + + if (filesForEmbedding.length > 0) { + let fileChunksEmbedded = 0 + let fileEmbedErrors = 0 + for (const file of filesForEmbedding) { + if ( + !isCurrentSourceVersion( + { sourcePath: file.path, sourceVersion: file.sourceVersion }, + logger, + ) + ) + continue + + try { + fileChunksEmbedded += await embedAndStoreFileChunks( + { + filePath: file.path, + sourceVersion: file.sourceVersion, + title: file.title, + content: file.content, + }, + logger, + ) + } catch (err) { + fileEmbedErrors++ + logger.warn("failed to embed file content", { + path: file.path, + error: describeError(err), + }) + } + + await setImmediateAsync() + } + logger.info("file content embedding pass complete", { + files: filesForEmbedding.length, + fileChunksEmbedded, + ...(fileEmbedErrors > 0 ? { fileEmbedErrors } : {}), + }) + } + + logger.info("embedding pass complete", { + notes: notesForEmbedding.length, + chunksEmbedded, + ...(entriesEmbedded > 0 ? { entriesEmbedded } : {}), + ...(embedErrors > 0 ? { embedErrors } : {}), + }) + })() + : Promise.resolve() // Log but don't crash on unhandled embedding errors embeddingPromise.catch((err) => { From a8821d8950dae593bd37585084180cf6213c75c1 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:11:44 -0400 Subject: [PATCH 04/13] fix(watcher): contain indexing and removal failures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6-astra --- .../search/__tests__/file-watcher.test.ts | 177 +++++++++++++++++- src/vault-mcp/search/file-watcher.ts | 54 +++++- 2 files changed, 225 insertions(+), 6 deletions(-) diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index 78df44910..fd81114d6 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -394,7 +394,7 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { watcher.emit("ready") await starting - const fire = async (event: "change" | "unlink", fileName: string): Promise => { + const fire = async (event: "add" | "change" | "unlink", fileName: string): Promise => { const handler = watcher.listeners(event)[0] if (!handler) throw new Error(`${event} handler was not registered`) @@ -405,6 +405,181 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { return { testVault, search, database, fire } } + it.each([{ event: "add" }, { event: "change" }] as const)( + "contains a non-markdown stat failure during $event and recovers", + async ({ event }) => { + const { testVault, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "image.png") + await writeFile(filePath, "image data") + const actualFs = + await vi.importActual("../../../utils/fs.js") + const failedPaths = new Set([filePath]) + const statSpy = vi.mocked(statOrNull).mockImplementation(async (requestedPath) => { + if (failedPaths.has(requestedPath)) throw new Error("controlled stat failure") + return actualFs.statOrNull(requestedPath) + }) + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => { + statSpy.mockRestore() + errorSpy.mockRestore() + }) + + await expect(fire(event, "image.png")).resolves.toBeUndefined() + expect(statSpy).toHaveBeenCalledWith(filePath) + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to stat non-md file", { + path: "image.png", + error: "[Error]: controlled stat failure", + }) + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + failedPaths.delete(filePath) + await fire(event, "image.png") + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([ + { path: "image.png" }, + ]) + expect(errorSpy).toHaveBeenCalledTimes(1) + }, + ) + + it.each([{ event: "add" }, { event: "change" }] as const)( + "contains an asset metadata upsert failure during $event and recovers", + async ({ event }) => { + const { testVault, search, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "content.txt") + await writeFile(filePath, "currentopal") + const realUpsert = search.upsertNonMdFile + const failedPaths = new Set(["content.txt"]) + const upsertSpy = vi + .spyOn(search, "upsertNonMdFile") + .mockImplementation((requestedPath, bytes) => { + if (failedPaths.has(requestedPath)) throw new Error("controlled asset upsert failure") + realUpsert(requestedPath, bytes) + }) + const contentUpsertSpy = vi.spyOn(search, "upsertFileContent") + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + + await expect(fire(event, "content.txt")).resolves.toBeUndefined() + expect(upsertSpy).toHaveBeenCalledExactlyOnceWith( + "content.txt", + Buffer.byteLength("currentopal"), + ) + expect(contentUpsertSpy).not.toHaveBeenCalled() + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to index non-md file metadata", { + path: "content.txt", + error: "[Error]: controlled asset upsert failure", + }) + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + failedPaths.delete("content.txt") + await fire(event, "content.txt") + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([ + { path: "content.txt" }, + ]) + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ + { path: "content.txt", content: "currentopal" }, + ]) + expect(errorSpy).toHaveBeenCalledTimes(1) + }, + ) + + it("contains an asset metadata removal failure during unlink and allows a later unlink", async () => { + const { testVault, search, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "content.txt") + await writeFile(filePath, "currentopal") + await fire("add", "content.txt") + await unlink(filePath) + const realRemove = search.removeNonMdFile + const removeSpy = vi.spyOn(search, "removeNonMdFile").mockImplementation((requestedPath) => { + if (requestedPath === "content.txt") throw new Error("controlled asset removal failure") + realRemove(requestedPath) + }) + const contentRemoveSpy = vi.spyOn(search, "removeFileContent") + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + + await expect(fire("unlink", "content.txt")).resolves.toBeUndefined() + expect(removeSpy).toHaveBeenCalledExactlyOnceWith("content.txt") + expect(contentRemoveSpy).not.toHaveBeenCalled() + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to remove non-md file metadata", { + path: "content.txt", + error: "[Error]: controlled asset removal failure", + }) + expect(await statOrNull(filePath)).toBeNull() + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([ + { path: "content.txt" }, + ]) + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ + { path: "content.txt", content: "currentopal" }, + ]) + removeSpy.mockRestore() + await fire("unlink", "content.txt") + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + expect(database.prepare("SELECT path FROM file_content").all()).toEqual([]) + expect(errorSpy).toHaveBeenCalledTimes(1) + }) + + it("contains a file content removal failure during unlink and allows a later unlink", async () => { + const { testVault, search, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "content.txt") + await writeFile(filePath, "currentopal") + await fire("add", "content.txt") + await unlink(filePath) + const realRemove = search.removeFileContent + const removeSpy = vi + .spyOn(search, "removeFileContent") + .mockImplementation((params, requestLogger) => { + if (params.filePath === "content.txt") throw new Error("controlled content removal failure") + realRemove(params, requestLogger) + }) + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + + await expect(fire("unlink", "content.txt")).resolves.toBeUndefined() + expect(removeSpy).toHaveBeenCalledExactlyOnceWith({ filePath: "content.txt" }, logger) + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to remove file content", { + path: "content.txt", + error: "[Error]: controlled content removal failure", + }) + expect(await statOrNull(filePath)).toBeNull() + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ + { path: "content.txt", content: "currentopal" }, + ]) + removeSpy.mockRestore() + await fire("unlink", "content.txt") + expect(database.prepare("SELECT path FROM file_content").all()).toEqual([]) + expect(errorSpy).toHaveBeenCalledTimes(1) + }) + + it("contains a note removal failure during unlink and allows a later unlink", async () => { + const { testVault, search, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "note.md") + await writeFile(filePath, "currentopal") + await fire("add", "note.md") + await unlink(filePath) + const realRemove = search.removeNote + const removeSpy = vi.spyOn(search, "removeNote").mockImplementation((requestedPath) => { + if (requestedPath === "note.md") throw new Error("controlled note removal failure") + realRemove(requestedPath) + }) + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + + await expect(fire("unlink", "note.md")).resolves.toBeUndefined() + expect(removeSpy).toHaveBeenCalledExactlyOnceWith("note.md") + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to remove note from index", { + path: "note.md", + error: "[Error]: controlled note removal failure", + }) + expect(await statOrNull(filePath)).toBeNull() + expect(database.prepare("SELECT path, content FROM notes").all()).toEqual([ + { path: "note.md", content: "currentopal" }, + ]) + removeSpy.mockRestore() + await fire("unlink", "note.md") + expect(database.prepare("SELECT path FROM notes").all()).toEqual([]) + expect(errorSpy).toHaveBeenCalledTimes(1) + }) + const delayFirstRead = async (filePath: string) => { const actualFs = await vi.importActual("node:fs/promises") const entered = Promise.withResolvers() diff --git a/src/vault-mcp/search/file-watcher.ts b/src/vault-mcp/search/file-watcher.ts index 10b9bb49c..38788ba72 100644 --- a/src/vault-mcp/search/file-watcher.ts +++ b/src/vault-mcp/search/file-watcher.ts @@ -5,6 +5,7 @@ import { watch } from "chokidar" import { DateTime } from "luxon" import { readFile, stat } from "node:fs/promises" +import type { Stats } from "node:fs" import { extname, join, relative, resolve as resolvePath } from "node:path" import { INDEXABLE_TEXT_EXTENSIONS } from "./search-index.js" import type { SearchIndex } from "./search-index.js" @@ -69,13 +70,32 @@ export const startFileWatcher = ( try { if (!filePath.endsWith(".md")) { - const fileStat = await statOrNull(filePath) + const readNonMdFileStat = async (): Promise => { + try { + return await statOrNull(filePath) + } catch (error) { + logger.error("failed to stat non-md file", { + path: relativePath, + error: describeError(error), + }) + return null + } + } + const fileStat = await readNonMdFileStat() if (currentEvents.get(relativePath) !== eventToken) return // Vanished between the watcher event and the stat — the unlink event // that follows will remove any existing row. if (!fileStat) return - search.upsertNonMdFile(relativePath, fileStat.size) + try { + search.upsertNonMdFile(relativePath, fileStat.size) + } catch (error) { + logger.error("failed to index non-md file metadata", { + path: relativePath, + error: describeError(error), + }) + return + } // Canvas files are always read — link extraction is unconditional. // PDF and text files are only read when file content FTS is enabled. @@ -209,20 +229,44 @@ export const startFileWatcher = ( currentEvents.delete(relativePath) if (!filePath.endsWith(".md")) { - search.removeNonMdFile(relativePath) + try { + search.removeNonMdFile(relativePath) + } catch (error) { + logger.error("failed to remove non-md file metadata", { + path: relativePath, + error: describeError(error), + }) + return + } const deletedExtension = extname(filePath) const isDeletedCanvas = deletedExtension === ".canvas" const isDeletedIndexable = search.fileContentIndexingEnabled && INDEXABLE_TEXT_EXTENSIONS.has(deletedExtension) if (isDeletedCanvas || isDeletedIndexable) { - search.removeFileContent({ filePath: relativePath }, logger) + try { + search.removeFileContent({ filePath: relativePath }, logger) + } catch (error) { + logger.error("failed to remove file content", { + path: relativePath, + error: describeError(error), + }) + return + } } logger.debug("removed non-md file from index", { path: relativePath }) return } - search.removeNote(relativePath) + try { + search.removeNote(relativePath) + } catch (error) { + logger.error("failed to remove note from index", { + path: relativePath, + error: describeError(error), + }) + return + } logger.debug("removed from index", { path: relativePath }) } From 9f8ebce563cbeb2efe40162ee462773244f85e32 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:25:09 -0400 Subject: [PATCH 05/13] test: strengthen embedding lifecycle regressions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: test-audit · gpt-6.1-sol --- .../search/__tests__/file-watcher.test.ts | 101 ++++++++++++++-- .../search/__tests__/hybrid-search.test.ts | 41 +++---- .../search/__tests__/memory-index.test.ts | 109 ++++++++++++++---- .../search/__tests__/search-index.test.ts | 57 +++------ 4 files changed, 211 insertions(+), 97 deletions(-) diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index fd81114d6..0b5aeac0d 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -59,16 +59,20 @@ const waitFor = async (check: () => boolean, timeoutMs = 8000, intervalMs = 100) * regression fails every attempt — retries only absorb the macOS race. */ const REAL_WATCHER_RETRY = { retry: 2 } -beforeEach(async () => { - vault = await mkdtemp(join(tmpdir(), "watcher-test-")) - index = createSearchIndex(":memory:") -}) +const registerSharedVaultHooks = (): void => { + beforeEach(async () => { + vault = await mkdtemp(join(tmpdir(), "watcher-test-")) + index = createSearchIndex(":memory:") + }) -afterEach(async () => { - await rm(vault, { recursive: true }) -}) + afterEach(async () => { + await rm(vault, { recursive: true }) + }) +} describe("file-watcher", REAL_WATCHER_RETRY, () => { + registerSharedVaultHooks() + it("indexes a new .md file", { timeout: 15000 }, async () => { await startFileWatcher(vault, index, { stabilityThreshold: 200, @@ -225,13 +229,13 @@ describe("file-watcher", REAL_WATCHER_RETRY, () => { await waitFor(() => embedNoteSpy.mock.calls.length > 0) const sourceVersion = await capturedVersion.promise - expect(embedNoteSpy).toHaveBeenCalledWith( + expect(embedNoteSpy).toHaveBeenCalledExactlyOnceWith( { notePath: "embed-test.md", rawContent: "---\ntitle: Embed\n---\n\nEmbed this content\n", sourceVersion, }, - expect.anything(), // logger — runtime child logger, not deterministic + logger, ) }) @@ -280,6 +284,8 @@ describe("file-watcher", REAL_WATCHER_RETRY, () => { }) describe("file-watcher — file content indexing", REAL_WATCHER_RETRY, () => { + registerSharedVaultHooks() + it("indexes a text file into file content FTS", { timeout: 15000 }, async () => { const fileIndex = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true, @@ -366,9 +372,9 @@ describe("file-watcher — file content indexing", REAL_WATCHER_RETRY, () => { await waitFor(() => embedFileSpy.mock.calls.length > 0) const sourceVersion = await capturedVersion.promise - expect(embedFileSpy).toHaveBeenCalledWith( + expect(embedFileSpy).toHaveBeenCalledExactlyOnceWith( { filePath: "data.csv", sourceVersion }, - expect.anything(), // logger — runtime child logger + logger, ) }) }) @@ -897,6 +903,75 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { }, ) + it("reports a detached file embedding failure and runs its queued replacement", async () => { + const failingEntered = Promise.withResolvers() + const releaseFailure = Promise.withResolvers() + const recoveredEntered = Promise.withResolvers() + const releaseRecovered = Promise.withResolvers() + const recoveredFinished = Promise.withResolvers() + const failureLogged = Promise.withResolvers() + const vector = new Float32Array(384).fill(0.1) + const embedder = { + embedText: vi.fn(async (text: string) => { + if (text === "content\n\noldquartz") { + failingEntered.resolve(undefined) + return releaseFailure.promise + } + if (text === "content\n\nnewopal") { + recoveredEntered.resolve(undefined) + return releaseRecovered.promise + } + return vector + }), + embedBatch: vi.fn(async (texts: readonly string[]) => texts.map(() => vector)), + } + const { testVault, database, search, fire } = await createControlledWatcher(embedder) + const realEmbed = search.embedFileContent + vi.spyOn(search, "embedFileContent").mockImplementation(async (params, requestLogger) => { + await realEmbed(params, requestLogger) + recoveredFinished.resolve(undefined) + }) + const warnSpy = vi.spyOn(logger, "warn").mockImplementation((message) => { + if (message === "file content embedding failed") failureLogged.resolve(undefined) + }) + const debugSpy = vi.spyOn(logger, "debug") + onTestFinished(() => { + warnSpy.mockRestore() + debugSpy.mockRestore() + }) + const filePath = join(testVault, "content.txt") + await writeFile(filePath, "oldquartz") + await fire("change", "content.txt") + await failingEntered.promise + await writeFile(filePath, "newopal") + await fire("change", "content.txt") + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ + { path: "content.txt", content: "newopal" }, + ]) + + releaseFailure.reject(new Error("controlled file model failure")) + await failureLogged.promise + await recoveredEntered.promise + expect(warnSpy).toHaveBeenCalledExactlyOnceWith("file content embedding failed", { + path: "content.txt", + error: "[Error]: controlled file model failure", + }) + expect(debugSpy).toHaveBeenCalledWith("previous file embed failed, proceeding with current", { + path: "content.txt", + error: "[Error]: controlled file model failure", + }) + expect(database.prepare("SELECT chunk_text FROM file_content_chunks").all()).toEqual([]) + releaseRecovered.resolve(vector) + await recoveredFinished.promise + expect(database.prepare("SELECT file_path, chunk_text FROM file_content_chunks").all()).toEqual( + [{ file_path: "content.txt", chunk_text: "content\n\nnewopal" }], + ) + expect(database.prepare("SELECT COUNT(*) AS count FROM file_content_vectors").get()).toEqual({ + count: 1, + }) + expect(embedder.embedText).toHaveBeenCalledTimes(2) + }) + it("recovers after a rejected job while another path embeds independently", async () => { const failingEntered = Promise.withResolvers() const releaseFailure = Promise.withResolvers() @@ -967,6 +1042,8 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { }) describe("startFileWatcher — chokidar watch options", () => { + registerSharedVaultHooks() + type FakeWatcher = { on: (event: string, handler: (...args: unknown[]) => void) => FakeWatcher } @@ -1031,6 +1108,8 @@ describe("startFileWatcher — chokidar watch options", () => { // captures the addDir handler, reports test-controlled tracking via // getWatched(), and records add() calls — against a real temp vault and index. describe("startFileWatcher — new-directory rescan", REAL_WATCHER_RETRY, () => { + registerSharedVaultHooks() + const RESCAN_TEST_OPTIONS = { stabilityThreshold: 200, pollInterval: 50, diff --git a/src/vault-mcp/search/__tests__/hybrid-search.test.ts b/src/vault-mcp/search/__tests__/hybrid-search.test.ts index d322d21ab..39dd679bb 100644 --- a/src/vault-mcp/search/__tests__/hybrid-search.test.ts +++ b/src/vault-mcp/search/__tests__/hybrid-search.test.ts @@ -348,8 +348,7 @@ Content about deployment costs and infrastructure. // Only the note inside Work/ should appear const paths = results.map((result) => result.path) - expect(paths).toContain("Work/inside.md") - expect(paths).not.toContain("Personal/outside.md") + expect(paths).toEqual(["Work/inside.md"]) }) it("matches the folder filter case-insensitively for vector-only results, like the FTS leg", async () => { @@ -436,8 +435,7 @@ Content about deployment costs and infrastructure. // Only c.md has the "work" tag — a.md (tags: personal, career) excluded const paths = results.map((result) => result.path) - expect(paths).toContain("c.md") - expect(paths).not.toContain("a.md") + expect(paths).toEqual(["c.md"]) }) it("applies type filter to vector-only results", async () => { @@ -469,8 +467,7 @@ Content about deployment costs and infrastructure. // Only c.md is type "meeting" — a.md (type: reflection) excluded const paths = results.map((result) => result.path) - expect(paths).toContain("c.md") - expect(paths).not.toContain("a.md") + expect(paths).toEqual(["c.md"]) }) it("applies created filter to vector-only results", async () => { @@ -753,8 +750,7 @@ The main content discusses RESTful API design and GraphQL alternatives. // Only c.md has the related link — a.md has no related field const paths = results.map((result) => result.path) - expect(paths).toContain("c.md") - expect(paths).not.toContain("a.md") + expect(paths).toEqual(["c.md"]) }) it("applies properties filter to vector-only results", async () => { @@ -822,8 +818,7 @@ This project is no longer maintained but had deployment infrastructure. // Only active.md has status: active const paths = results.map((result) => result.path) - expect(paths).toContain("active.md") - expect(paths).not.toContain("archived.md") + expect(paths).toEqual(["active.md"]) }) }) @@ -988,14 +983,10 @@ This is a note with many words that should be truncated when using a small snipp await rerankIndex.hybridSearch({ query: "career goals" }, logger) - expect(mockReranker.rerankPairs).toHaveBeenCalledOnce() - const callArgs = mockReranker.rerankPairs.mock.calls[0] - expect(callArgs).toBeDefined() - const [query, documents] = callArgs ?? [] - expect(query).toBe("career goals") - expect(documents).toHaveLength(2) - // Each document text should be non-empty (chunk text from vector hits) - expect(documents.every((document: string) => document.length > 0)).toBe(true) + expect(mockReranker.rerankPairs).toHaveBeenCalledExactlyOnceWith("career goals", [ + "Career Goals\n\n\nI aspire to build meaningful products and grow as a technical leader.\nMy targets include shipping a major open source project.", + "Project Ideas\n\n\nSome project ideas for the next quarter. Build a CLI tool for vault management.", + ]) }) it("modifies result scores compared to RRF-only ordering", async () => { @@ -1146,9 +1137,10 @@ describe("hybridSearch — file content vector search", () => { expect(search_mode).toBe("hybrid") // Both should appear — guide.txt via FTS+file vector, career.md via note vector - expect(results.map((result) => result.path)).toEqual(["docs/guide.txt", "notes/career.md"]) - expect(results[0]?.kind).toBe("file") - expect(results[0]?.extension).toBe(".txt") + expect(results.map(({ path, kind, extension }) => ({ path, kind, extension }))).toEqual([ + { path: "docs/guide.txt", kind: "file", extension: ".txt" }, + { path: "notes/career.md", kind: "note", extension: undefined }, + ]) }) it("file-only vector hit appears with metadata when no FTS match exists", async () => { @@ -1182,10 +1174,9 @@ describe("hybridSearch — file content vector search", () => { // Exactly one result — the file via vector-only (no FTS match) expect(search_mode).toBe("hybrid") - expect(results.map((result) => result.path)).toEqual(["specs/api-spec.yaml"]) - expect(results[0]?.kind).toBe("file") - expect(results[0]?.extension).toBe(".yaml") - expect(results[0]?.folder).toBe("specs") + expect( + results.map(({ path, kind, extension, folder }) => ({ path, kind, extension, folder })), + ).toEqual([{ path: "specs/api-spec.yaml", kind: "file", extension: ".yaml", folder: "specs" }]) }) it("returns note results when the file content KNN query throws", async () => { diff --git a/src/vault-mcp/search/__tests__/memory-index.test.ts b/src/vault-mcp/search/__tests__/memory-index.test.ts index 96fb39824..0aaa74005 100644 --- a/src/vault-mcp/search/__tests__/memory-index.test.ts +++ b/src/vault-mcp/search/__tests__/memory-index.test.ts @@ -21,11 +21,9 @@ const DIMENSIONS = 384 /** Creates a mock embedder that returns deterministic embeddings. */ const createMockEmbedder = () => ({ embedText: vi.fn().mockResolvedValue(new Float32Array(DIMENSIONS).fill(0.1)), - embedBatch: vi - .fn() - .mockImplementation((texts: string[]) => - Promise.resolve(texts.map(() => new Float32Array(DIMENSIONS).fill(0.1))), - ), + embedBatch: vi.fn(async (texts: readonly string[]): Promise => { + return texts.map(() => new Float32Array(DIMENSIONS).fill(0.1)) + }), }) /** Builds a fileStat object for upsertNote. Defaults to size 100. */ @@ -37,10 +35,7 @@ const testStat = (mtimeMs: number, size = 100): { mtimeMs: number; size: number /** Total entry texts sent to the embedder across all embedBatch calls — * the observable that proves how many entries were actually (re-)embedded. */ const totalTextsEmbedded = (embedder: ReturnType): number => - embedder.embedBatch.mock.calls.reduce( - (sum: number, call: unknown[]) => sum + (call[0] as string[]).length, - 0, - ) + embedder.embedBatch.mock.calls.reduce((sum, [texts]) => sum + texts.length, 0) /** File-backed index plus a second read-only connection for asserting raw * table state — :memory: databases can't be inspected from outside the @@ -183,10 +178,7 @@ describe("memory entry indexing", () => { { sourceVersion: sourceVersion, notePath: "About Me/Opinions.md", rawContent: OPINIONS_V1 }, logger, ) - const embeddedEntryTexts = embedder.embedBatch.mock.calls.flatMap( - (call: unknown[]) => call[0] as string[], - ) - expect(embeddedEntryTexts).toEqual([ + expect(embedder.embedBatch).toHaveBeenCalledExactlyOnceWith([ "Opinions > Code patterns (newest first)\n- **2026-07-02**: Wrap function bodies in braces.", "Opinions > Code patterns (newest first)\n- **2026-05-07**: Immutable over mutable.", "Opinions > Process (newest first)\n- **2026-06-25**: Sequential over parallel review.", @@ -238,10 +230,7 @@ describe("memory entry indexing", () => { logger, ) - expect(totalTextsEmbedded(embedder)).toBe(1) - expect( - embedder.embedBatch.mock.calls.flatMap((call: unknown[]) => call[0] as string[]), - ).toEqual([ + expect(embedder.embedBatch).toHaveBeenCalledExactlyOnceWith([ "Opinions > Code patterns (newest first)\n- **2026-07-11**: Newest opinion lands on top.", ]) // The shifted entries kept their rows; indices were refreshed in place. @@ -292,7 +281,9 @@ describe("memory entry indexing", () => { logger, ) - expect(totalTextsEmbedded(embedder)).toBe(1) + expect(embedder.embedBatch).toHaveBeenCalledExactlyOnceWith([ + "Opinions > Code patterns (newest first)\n- **2026-05-07**: Immutable over mutable, always.", + ]) // Still 3 rows and 3 vectors — the old row and its vector are gone, not // orphaned beside the new ones. expect(selectEntryRows(inspect)).toHaveLength(3) @@ -405,8 +396,7 @@ describe("memory entry indexing", () => { ) const rows = selectEntryRows(inspect) - expect(rows).toHaveLength(3) - expect(rows.every((row) => row.file === "Beliefs")).toBe(true) + expect(rows.map((row) => row.file)).toEqual(["Beliefs", "Beliefs", "Beliefs"]) // One-time full re-embed under the new name — the documented rename cost. expect(totalTextsEmbedded(embedder)).toBe(3) }) @@ -459,6 +449,85 @@ describe("memory entry indexing", () => { }) describe("memory embedding source versions", () => { + it("keeps a committed first batch while rejecting a superseded second batch and recovers", async () => { + const { index, embedder, inspect } = await createInspectableMemoryIndex() + + if (!embedder) throw new Error("embedder required") + const entryTexts = Array.from({ length: 17 }, (_, entryIndex) => { + return `- **2026-07-01**: Practice ${String(entryIndex)} improves reliability.` + }) + const content = `## Practices\n\n${entryTexts.join("\n")}` + const secondBatchStarted = Promise.withResolvers() + const secondBatchModel = Promise.withResolvers() + embedder.embedBatch.mockImplementationOnce(async (texts) => { + return texts.map(() => new Float32Array(DIMENSIONS).fill(0.1)) + }) + embedder.embedBatch.mockImplementationOnce(() => { + secondBatchStarted.resolve(undefined) + return secondBatchModel.promise + }) + const originalVersion = index.upsertNote( + { filePath: "About Me/Opinions.md", rawContent: content, fileStat: testStat(1000) }, + logger, + ) + const staleJob = index.embedNote( + { notePath: "About Me/Opinions.md", rawContent: content, sourceVersion: originalVersion }, + logger, + ) + await secondBatchStarted.promise + const firstBatchVectors = inspect + .prepare( + "SELECT entry_id, hex(embedding) AS embedding FROM memory_entry_vectors ORDER BY entry_id", + ) + .all() + expect(firstBatchVectors).toHaveLength(16) + const replacementText = "- **2026-07-01**: Replacement practice improves recovery." + const replacementContent = `## Practices\n\n${[...entryTexts.slice(0, 16), replacementText].join("\n")}` + const replacementVersion = index.upsertNote( + { + filePath: "About Me/Opinions.md", + rawContent: replacementContent, + fileStat: testStat(1000), + }, + logger, + ) + secondBatchModel.resolve([new Float32Array(DIMENSIONS).fill(0.9)]) + await staleJob + + expect(embedder.embedBatch).toHaveBeenCalledTimes(2) + expect( + inspect + .prepare( + "SELECT entry_id, hex(embedding) AS embedding FROM memory_entry_vectors ORDER BY entry_id", + ) + .all(), + ).toEqual(firstBatchVectors) + expect(selectEntryRows(inspect).map((row) => row.entry_text)).toEqual([ + ...entryTexts.slice(0, 16), + replacementText, + ]) + embedder.embedBatch.mockClear() + await index.embedNote( + { + notePath: "About Me/Opinions.md", + rawContent: replacementContent, + sourceVersion: replacementVersion, + }, + logger, + ) + expect(embedder.embedBatch).toHaveBeenCalledExactlyOnceWith([ + `Opinions > Practices\n${replacementText}`, + ]) + expect(countVectors(inspect)).toBe(17) + expect( + inspect + .prepare( + "SELECT entry_id, hex(embedding) AS embedding FROM memory_entry_vectors ORDER BY entry_id LIMIT 16", + ) + .all(), + ).toEqual(firstBatchVectors) + }) + it.each(["delete", "recreate", "prune"] as const)( "stops an obsolete 17-entry batch after a source %s", async (mutation) => { diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index b67784818..caae5c7a9 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -5283,48 +5283,23 @@ It has multiple sentences to verify chunking works correctly. }) it("removeNote deletes associated chunks and vectors", async () => { - const mockEmbedder = createMockEmbedder() - const embeddingIndex = createSearchIndex(":memory:", mockEmbedder) - - const originalSourceVersion = embeddingIndex.upsertNote( - { - filePath: "test.md", - rawContent: NOTE_FOR_EMBEDDING, - fileStat: testStat(1000), - }, - logger, - ) - await embeddingIndex.embedNote( - { - sourceVersion: originalSourceVersion, - notePath: "test.md", - rawContent: NOTE_FOR_EMBEDDING, - }, - logger, - ) + const fixture = await createEmbeddingRaceIndex("note") + const originalSourceVersion = fixture.upsert(NOTE_FOR_EMBEDDING) + await fixture.embed(NOTE_FOR_EMBEDDING, originalSourceVersion) + expect(fixture.chunks()).toHaveLength(1) + expect(fixture.vectorCount()).toBe(1) - // Remove should not throw — cleanup should succeed - embeddingIndex.removeNote("test.md") + fixture.remove() + expect(fixture.inspect.prepare("SELECT path FROM notes").all()).toEqual([]) + expect(fixture.chunks()).toEqual([]) + expect(fixture.vectorCount()).toBe(0) - // Re-embedding after removal should embed again (not skip via hash) - mockEmbedder.embedText.mockClear() - const updatedSourceVersion = embeddingIndex.upsertNote( - { - filePath: "test.md", - rawContent: NOTE_FOR_EMBEDDING, - fileStat: testStat(2000), - }, - logger, - ) - await embeddingIndex.embedNote( - { - sourceVersion: updatedSourceVersion, - notePath: "test.md", - rawContent: NOTE_FOR_EMBEDDING, - }, - logger, - ) - expect(mockEmbedder.embedText).toHaveBeenCalled() + fixture.embedder.embedText.mockClear() + const updatedSourceVersion = fixture.upsert(NOTE_FOR_EMBEDDING) + await fixture.embed(NOTE_FOR_EMBEDDING, updatedSourceVersion) + expect(fixture.embedder.embedText).toHaveBeenCalledTimes(1) + expect(fixture.chunks()).toHaveLength(1) + expect(fixture.vectorCount()).toBe(1) }) it("embedNote produces a chunk even for empty content", async () => { @@ -5343,7 +5318,7 @@ It has multiple sentences to verify chunking works correctly. // chunker returns at least one chunk (the title-only fallback), so // embedText is called even for empty content - expect(mockEmbedder.embedText).toHaveBeenCalled() + expect(mockEmbedder.embedText).toHaveBeenCalledExactlyOnceWith("empty") }) it("embedNote propagates embedder errors to the caller", async () => { From d5ff2da40024c8884f1a9ca24bbd688ecd0fe470 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 21:40:30 -0400 Subject: [PATCH 06/13] fix: index bootstrapped memory and bound rebuild work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: bug-check · gpt-6.1-sol --- ARCHITECTURE.md | 7 +- .../integration/server-integration.test.ts | 47 ++++++- .../search/__tests__/file-watcher.test.ts | 6 +- .../search/__tests__/search-index.test.ts | 56 +++++++++ src/vault-mcp/search/file-watcher.ts | 2 +- src/vault-mcp/search/search-index.ts | 117 +++++++++--------- src/vault-mcp/server.ts | 2 +- 7 files changed, 167 insertions(+), 70 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index f6cd239a1..ad13967a2 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -897,10 +897,11 @@ graph LR `/home/obsidian/.config` (persists across restarts for incremental sync — critical for embedding ingestion). 3. **`svc-vault-mcp`** — MCP server. Drops to the same `obsidian` user, so - both processes read/write the shared `/vault` volume. On startup: builds - the FTS5 search index, bootstraps memory templates if the memory folder + both processes read/write the shared `/vault` volume. On startup: bootstraps + memory templates if the memory folder doesn't exist, `MEMORY_ENABLED` is not `false`, and the server is not in - `READONLY_MODE`, then starts the file watcher. + `READONLY_MODE`, builds the FTS5 search index including those templates, + then starts the file watcher. `svc-vault-mcp` declares `svc-obsidian-sync` in its `dependencies.d`, so the MCP server starts only after the full init chain has finished and the sync diff --git a/src/__tests__/integration/server-integration.test.ts b/src/__tests__/integration/server-integration.test.ts index a4db4620b..79a6c26c4 100644 --- a/src/__tests__/integration/server-integration.test.ts +++ b/src/__tests__/integration/server-integration.test.ts @@ -12,7 +12,7 @@ import { describe, it, expect, beforeAll, afterAll, onTestFinished, vi } from "vitest" import { DateTime } from "luxon" -import { readFile, stat, writeFile } from "node:fs/promises" +import { readFile, readdir, stat, writeFile } from "node:fs/promises" import { join } from "node:path" import Database from "better-sqlite3" import { fileExists } from "../../utils/fs.js" @@ -32,6 +32,51 @@ import type { ToolResult } from "./test-harness.js" vi.setConfig({ testTimeout: 15_000 }) +it("indexes newly bootstrapped memory templates before accepting requests", async () => { + const server = await startServer(await freePort(), { MEMORY_DIR: "Fresh Memory" }) + onTestFinished(server.cleanup) + const client = await createTestClient(server.port) + onTestFinished(() => client.close()) + const database = new Database(join(server.dataDir, "search.db"), { readonly: true }) + onTestFinished(() => { + database.close() + }) + + expect((await readdir(join(server.vaultPath, "Fresh Memory"))).toSorted()).toEqual([ + "Agents.md", + "Me.md", + "Opinions.md", + "Principles.md", + "Routines.md", + ]) + const result = await callTool({ + client, + name: "vault_search", + args: { query: '"subject of every entry"', filters: { folder: "Fresh Memory" } }, + }) + const parsedResult = JSON.parse(textContent(result)) + + expect(parsedResult.results.map((entry: { path: string }) => entry.path)).toEqual([ + "Fresh Memory/Agents.md", + ]) + expect( + database.prepare("SELECT path FROM notes WHERE path LIKE 'Fresh Memory/%' ORDER BY path").all(), + ).toEqual([ + { path: "Fresh Memory/Agents.md" }, + { path: "Fresh Memory/Me.md" }, + { path: "Fresh Memory/Opinions.md" }, + { path: "Fresh Memory/Principles.md" }, + { path: "Fresh Memory/Routines.md" }, + ]) + const expectedPreferences = await readFile( + join(import.meta.dirname, "fixtures/vault/About Me/Preferences.md"), + "utf8", + ) + expect(await readFile(join(server.vaultPath, "About Me/Preferences.md"), "utf8")).toBe( + expectedPreferences, + ) +}) + /** Extract joined text from a prompt result's messages. */ const promptText = (result: Awaited>): string => { return result.messages diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index 0b5aeac0d..99b548929 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -931,12 +931,12 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { await realEmbed(params, requestLogger) recoveredFinished.resolve(undefined) }) - const warnSpy = vi.spyOn(logger, "warn").mockImplementation((message) => { + const errorSpy = vi.spyOn(logger, "error").mockImplementation((message) => { if (message === "file content embedding failed") failureLogged.resolve(undefined) }) const debugSpy = vi.spyOn(logger, "debug") onTestFinished(() => { - warnSpy.mockRestore() + errorSpy.mockRestore() debugSpy.mockRestore() }) const filePath = join(testVault, "content.txt") @@ -952,7 +952,7 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { releaseFailure.reject(new Error("controlled file model failure")) await failureLogged.promise await recoveredEntered.promise - expect(warnSpy).toHaveBeenCalledExactlyOnceWith("file content embedding failed", { + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("file content embedding failed", { path: "content.txt", error: "[Error]: controlled file model failure", }) diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index caae5c7a9..165d9ba84 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -4948,6 +4948,62 @@ describe("file targets written with extensions", () => { expect(index.brokenLinkCount({}, logger).count).toBe(0) }) + it.each([ + { label: "full path", target: "assets/photo.png", filePath: "assets/photo.png" }, + { label: "relative path", target: "../assets/photo.png", filePath: "assets/photo.png" }, + { label: "filename suffix", target: "photo.png", filePath: "deep/assets/photo.png" }, + { label: "folded suffix", target: "PHOTO.png", filePath: "deep/assets/photo.png" }, + { label: "stem path", target: "assets/Route", filePath: "assets/Route.canvas" }, + { label: "relative stem", target: "../assets/Route", filePath: "assets/Route.canvas" }, + { label: "stem suffix", target: "assets/Route", filePath: "deep/assets/Route.canvas" }, + { label: "folded stem suffix", target: "assets/ROUTE", filePath: "deep/assets/Route.canvas" }, + { label: "bare stem", target: "Route", filePath: "assets/Route.canvas" }, + { label: "multi-dot stem", target: "photo.png", filePath: "assets/photo.png.canvas" }, + { label: "literal wildcards", target: "photo_%25.png", filePath: "assets/photo_%25.png" }, + ])( + "re-resolves a $label forward asset link past unrelated candidates", + ({ target, filePath }) => { + const assetIndex = createSearchIndex(":memory:") + assetIndex.upsertNote( + { + filePath: "Projects/source.md", + rawContent: `![[${target}]]\n![[unrelated.png]]`, + fileStat: testStat(1000), + }, + logger, + ) + assetIndex.upsertNonMdFile("elsewhere/decoy.png", 42) + expect( + assetIndex + .getOutgoingLinks({ path: "Projects/source.md" }, logger) + .map((link) => link.path), + ).toEqual([target, "unrelated.png"].toSorted()) + + assetIndex.upsertNonMdFile(filePath, 100) + + expect(assetIndex.getOutgoingLinks({ path: "Projects/source.md" }, logger)).toEqual( + [ + { + path: filePath, + title: null, + exists: true, + kind: "file", + bytes: 100, + daily_note_forward_ref: false, + }, + { + path: "unrelated.png", + title: null, + exists: false, + kind: "note", + bytes: null, + daily_note_forward_ref: false, + }, + ].toSorted((a, b) => a.path.localeCompare(b.path)), + ) + }, + ) + it("does not let LIKE wildcards in the target match unrelated files via full-path suffix", () => { // Only photo1final.png exists — if the _ in the target were treated as a // LIKE wildcard it would match (1 satisfies _), giving a false resolution. diff --git a/src/vault-mcp/search/file-watcher.ts b/src/vault-mcp/search/file-watcher.ts index 38788ba72..9e617e94e 100644 --- a/src/vault-mcp/search/file-watcher.ts +++ b/src/vault-mcp/search/file-watcher.ts @@ -147,7 +147,7 @@ export const startFileWatcher = ( * file handler returns; an older job must leave a newer queued job registered. */ currentEmbed .catch((embedError) => { - logger.warn("file content embedding failed", { + logger.error("file content embedding failed", { path: relativePath, error: describeError(embedError), }) diff --git a/src/vault-mcp/search/search-index.ts b/src/vault-mcp/search/search-index.ts index 375dff1b7..85d0927c8 100644 --- a/src/vault-mcp/search/search-index.ts +++ b/src/vault-mcp/search/search-index.ts @@ -310,7 +310,7 @@ export type OutgoingLinkEntry = { // Links between notes are tracked in a `links` table (source → target) // to power backlink queries, outgoing link lookups, and orphan detection. // The link grammar — recognizing, parsing, and resolving links — lives in -// ../links.ts; this section only composes it for indexing. +// ../obsidian-markdown/links.ts; this section only composes it for indexing. // // Indexing flow: // 1. links.extractFromBody() parses wikilinks ([[target]]) and markdown @@ -616,8 +616,8 @@ export const createSearchIndex = ( } // Prepared statements are compiled once here and reused across all calls. - // db.prepare() caches the compiled SQL — calling it inside a function - // would re-compile on every invocation. + // Each returned statement retains its compiled SQL; preparing inside an + // operation would compile it again on every invocation. const upsertNotesStmt = db.prepare(` INSERT OR REPLACE INTO notes (path, title, content, tags, related, folder, type, created, mtime, properties, leading_callout, bytes, kanban_done_lanes) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) @@ -1157,8 +1157,28 @@ export const createSearchIndex = ( // Re-resolve unresolved links that now match this non-md file — upgrade // raw targets (e.g. "Trip Route") to resolved paths ("Trip Route.canvas"). + const foldedFilePath = filePath.toLowerCase() + const foldedBasePath = basePath.toLowerCase() + const matchesNewFile = (link: { source: string; target: string }): boolean => { + const relativeTarget = posix.join(posix.dirname(link.source), link.target) + const targetSuffix = `/${link.target.toLowerCase()}` + + // This broad suffix check includes SQLite LIKE's ASCII case folding; + // the resolver remains authoritative for match ordering and exact case. + return ( + filePath === link.target || + filePath === relativeTarget || + foldedFilePath.endsWith(targetSuffix) || + basePath === link.target || + basePath === relativeTarget || + foldedBasePath.endsWith(targetSuffix) || + baseFilename === link.target + ) + } const unresolvedLinks = selectUnresolvedLinksStmt.all() for (const link of unresolvedLinks) { + if (!matchesNewFile(link)) continue + const resolvedPath = resolveNonMarkdownFile({ target: link.target, sourcePath: link.source }) if (resolvedPath !== null) { @@ -2076,9 +2096,14 @@ export const createSearchIndex = ( if (!fileStat) return null return { relativePath: file.relativePath, bytes: fileStat.size } } - const nonMarkdownFileSizes = (await Promise.all(allNonMdFiles.map(readFileSize))).filter( - (entry) => entry !== null, - ) + /** Bound filesystem work so a large vault cannot open every file at once. */ + const REBUILD_IO_CONCURRENCY = 16 + const nonMarkdownFileSizeResults = await mapWithConcurrency({ + items: allNonMdFiles, + concurrency: REBUILD_IO_CONCURRENCY, + mapper: readFileSize, + }) + const nonMarkdownFileSizes = nonMarkdownFileSizeResults.filter((entry) => entry !== null) const canvasFiles = allNonMdFiles.filter((file) => file.relativePath.endsWith(".canvas")) // PDF and text files are only read when file content FTS is enabled — // without the tables, the extraction is wasted I/O. @@ -2093,9 +2118,12 @@ export const createSearchIndex = ( : [] // Read canvas files for content indexing + link extraction. - const readCanvasContent = async ( - file: RebuildFilePaths, - ): Promise => { + const readRebuildFileContent = async (params: { + file: RebuildFilePaths + sourceKind: "canvas file" | "text file" | "note" + }): Promise => { + const { file, sourceKind } = params + try { const [content, fileStat] = await Promise.all([ readFile(file.absolutePath, "utf8"), @@ -2108,16 +2136,19 @@ export const createSearchIndex = ( sizeBytes: fileStat.size, } } catch (error) { - logger.warn("skipped unreadable canvas file during rebuild", { + logger.warn(`skipped unreadable ${sourceKind} during rebuild`, { path: file.relativePath, error: describeError(error), }) return null } } - const canvasContents = (await Promise.all(canvasFiles.map(readCanvasContent))).filter( - (entry) => entry !== null, - ) + const canvasContentResults = await mapWithConcurrency({ + items: canvasFiles, + concurrency: REBUILD_IO_CONCURRENCY, + mapper: (file) => readRebuildFileContent({ file, sourceKind: "canvas file" }), + }) + const canvasContents = canvasContentResults.filter((entry) => entry !== null) // Extract PDF text with bounded concurrency (CPU-intensive pdfjs work). const extractPdfContent = async (file: { @@ -2158,55 +2189,19 @@ export const createSearchIndex = ( const pdfContents = pdfResults.filter((entry) => entry !== null) // Read text files for content indexing (raw UTF-8). - const readTextFileContent = async ( - file: RebuildFilePaths, - ): Promise => { - try { - const [content, fileStat] = await Promise.all([ - readFile(file.absolutePath, "utf8"), - stat(file.absolutePath), - ]) - return { - relativePath: file.relativePath, - content, - modifiedAtMs: fileStat.mtimeMs, - sizeBytes: fileStat.size, - } - } catch (error) { - logger.warn("skipped unreadable text file during rebuild", { - path: file.relativePath, - error: describeError(error), - }) - return null - } - } - const textFileContents = (await Promise.all(textFiles.map(readTextFileContent))).filter( - (entry) => entry !== null, - ) + const textFileContentResults = await mapWithConcurrency({ + items: textFiles, + concurrency: REBUILD_IO_CONCURRENCY, + mapper: (file) => readRebuildFileContent({ file, sourceKind: "text file" }), + }) + const textFileContents = textFileContentResults.filter((entry) => entry !== null) - const readNoteContent = async (file: RebuildFilePaths): Promise => { - try { - const [content, fileStat] = await Promise.all([ - readFile(file.absolutePath, "utf8"), - stat(file.absolutePath), - ]) - return { - relativePath: file.relativePath, - content, - modifiedAtMs: fileStat.mtimeMs, - sizeBytes: fileStat.size, - } - } catch (error) { - logger.warn("skipped unreadable note during rebuild", { - path: file.relativePath, - error: describeError(error), - }) - return null - } - } - const noteContents = (await Promise.all(markdownFiles.map(readNoteContent))).filter( - (entry) => entry !== null, - ) + const noteContentResults = await mapWithConcurrency({ + items: markdownFiles, + concurrency: REBUILD_IO_CONCURRENCY, + mapper: (file) => readRebuildFileContent({ file, sourceKind: "note" }), + }) + const noteContents = noteContentResults.filter((entry) => entry !== null) // Notes whose parse or index write throws are skipped with a warning // instead of aborting the rebuild — one malformed note must never diff --git a/src/vault-mcp/server.ts b/src/vault-mcp/server.ts index 0e2216b2e..4735eb9cd 100644 --- a/src/vault-mcp/server.ts +++ b/src/vault-mcp/server.ts @@ -146,10 +146,10 @@ const startServer = async (): Promise => { fileToolsEnabled: config.fileToolsEnabled, statusRegistry, }) + await bootstrapMemoryIfEnabled(config, vaultPath) const { count } = await search.rebuildFromVault({ vaultPath }, logger) logger.info("initial index built", { count }) - await bootstrapMemoryIfEnabled(config, vaultPath) await startFileWatcher(vaultPath, search, { usePolling: config.windowsBindMount, }) From dfd74a0b4aff06d37154349aea60da4032dd31f2 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:03:22 -0400 Subject: [PATCH 07/13] docs: align startup guides with template indexing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6-astra --- ARCHITECTURE.md | 2 +- deploy/local/README.md | 14 +++++++++----- deploy/remote/README.md | 16 ++++++++++------ 3 files changed, 20 insertions(+), 12 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index ad13967a2..4da30d87b 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -117,7 +117,7 @@ graph TB **Hybrid query:** MCP client → `vault_search` → FTS5 BM25 ranks (notes + file content) + sqlite-vec KNN ranks (notes + file content) → RRF fusion → cross-encoder reranking → response. -**Invariant — vault is source of truth:** Vault files are canonical. MCP content edits write to those files, and the watcher and startup rebuild derive the SQLite search index from them. +**Invariant — vault is source of truth:** Vault files are canonical. MCP content edits must write to those files, never directly to the index. The watcher and startup rebuild derive the SQLite search index from the files. Embeddings persist across rebuilds to reuse unchanged content and can be regenerated from the vault. diff --git a/deploy/local/README.md b/deploy/local/README.md index ef23fcbcc..2cee67902 100644 --- a/deploy/local/README.md +++ b/deploy/local/README.md @@ -188,11 +188,15 @@ docker compose pull && docker compose up -d ## Restart -The server runs startup tasks on every boot: it rebuilds the search index, -creates memory template files if the memory folder doesn't exist (skipped -when `MEMORY_ENABLED=false` or `READONLY_MODE=true`), and starts -the file watcher. Restarting the container re-runs this flow (useful when -testing bootstrap behavior). The command is the same for both setup methods, +The server runs these startup tasks on every boot: + +1. Create memory template files if the memory folder doesn't exist (skipped + when `MEMORY_ENABLED=false` or `READONLY_MODE=true`). +2. Rebuild the search index, including any new memory template files. +3. Start the file watcher. + +Restarting the container re-runs this flow (useful when testing bootstrap +behavior). The command is the same for both setup methods, since both name the container `vault-cortex`: ```bash diff --git a/deploy/remote/README.md b/deploy/remote/README.md index adccbcc33..d86b86bf8 100644 --- a/deploy/remote/README.md +++ b/deploy/remote/README.md @@ -432,12 +432,16 @@ and unchanged notes are not re-embedded. ## Restart -The container runs startup tasks on every boot: a catch-up sync runs before -the server starts to bring the vault current, then the server rebuilds the -search index, creates memory template files if the memory folder doesn't -exist (skipped when `MEMORY_ENABLED=false` or `READONLY_MODE=true`), and -starts the file watcher. Restarting the container re-runs this flow (useful -when testing bootstrap behavior). The command is the same for every setup +On every boot, a catch-up sync brings the vault current before the server +starts. The server then runs these startup tasks: + +1. Create memory template files if the memory folder doesn't exist (skipped + when `MEMORY_ENABLED=false` or `READONLY_MODE=true`). +2. Rebuild the search index, including any new memory template files. +3. Start the file watcher. + +Restarting the container re-runs this flow (useful when testing bootstrap +behavior). The command is the same for every setup method: ```bash From 1101a6a85e764d01120e5a841066294a328a3e98 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:30:38 -0400 Subject: [PATCH 08/13] fix(memory): bound simultaneous file reads MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6-astra --- .../__tests__/memory-store.test.ts | 74 +++++++++++++++++++ .../vault-operations/memory-store.ts | 21 ++++-- 2 files changed, 87 insertions(+), 8 deletions(-) diff --git a/src/vault-mcp/vault-operations/__tests__/memory-store.test.ts b/src/vault-mcp/vault-operations/__tests__/memory-store.test.ts index f5798af90..a7de280b1 100644 --- a/src/vault-mcp/vault-operations/__tests__/memory-store.test.ts +++ b/src/vault-mcp/vault-operations/__tests__/memory-store.test.ts @@ -15,6 +15,51 @@ import { tmpdir } from "node:os" import { parseNote } from "../../obsidian-markdown/frontmatter.js" import { createMemoryStore } from "../memory-store.js" import { logger } from "../../../logger.js" +vi.mock("node:fs/promises", { spy: true }) + +const createBlockedMemoryReads = async (vaultPath: string) => { + const memoryDir = "Bounded" + await mkdir(join(vaultPath, memoryDir)) + const files = Array.from({ length: 17 }, (_unused, index) => { + const name = `Entry-${index.toString().padStart(2, "0")}` + return { name, content: index === 16 ? "" : `# ${name}` } + }) + const fileNames = files.map((file) => file.name) + const fileContents = files.map((file) => file.content) + await Promise.all( + files.map((file) => { + return writeFile(join(vaultPath, memoryDir, `${file.name}.md`), file.content) + }), + ) + const contentByPath = new Map( + files.map((file) => { + return [join(vaultPath, memoryDir, `${file.name}.md`), file.content] as const + }), + ) + const firstBatchStarted = Promise.withResolvers() + const releaseReads = Promise.withResolvers() + const startedReads: string[] = [] + onTestFinished(() => { + releaseReads.resolve(undefined) + vi.mocked(readFile).mockRestore() + }) + vi.mocked(readFile).mockImplementation(async (filePath) => { + const content = contentByPath.get(String(filePath)) + + if (content === undefined) throw new Error("unexpected memory fixture path") + startedReads.push(String(filePath)) + if (startedReads.length === 16) firstBatchStarted.resolve(undefined) + await releaseReads.promise + return content + }) + return { + store: createMemoryStore({ memoryDir }), + fileNames, + fileContents, + firstBatchStarted: firstBatchStarted.promise, + releaseReads, + } +} const { getMemory, @@ -85,6 +130,20 @@ afterEach(async () => { }) describe("getMemory", () => { + it("bounds simultaneous all-file reads and retains every file in order", async () => { + const fixture = await createBlockedMemoryReads(vault) + const reading = fixture.store.getMemory({ vaultPath: vault }, logger) + onTestFinished(async () => { + fixture.releaseReads.resolve(undefined) + await reading + }) + await fixture.firstBatchStarted + expect(readFile).toHaveBeenCalledTimes(16) + fixture.releaseReads.resolve(undefined) + expect(await reading).toBe(fixture.fileContents.join("\n\n---\n\n")) + expect(readFile).toHaveBeenCalledTimes(17) + }) + // A pre-existing hidden file on disk (created outside the server) must not // leak through the concatenate-all read — the enumeration filter, not just // the explicit-name rejection, is what excludes it. @@ -1779,6 +1838,21 @@ created: 2026-01-01T00:00:00-05:00 }) describe("listMemoryFiles", () => { + it("bounds simultaneous outline reads and retains every file in order", async () => { + const fixture = await createBlockedMemoryReads(vault) + const reading = fixture.store.listMemoryFiles({ vaultPath: vault }, logger) + onTestFinished(async () => { + fixture.releaseReads.resolve(undefined) + await reading + }) + await fixture.firstBatchStarted + expect(readFile).toHaveBeenCalledTimes(16) + fixture.releaseReads.resolve(undefined) + const outlines = await reading + expect(outlines.map((outline) => outline.file)).toEqual(fixture.fileNames) + expect(readFile).toHaveBeenCalledTimes(17) + }) + it("excludes a pre-existing hidden memory file from the outlines", async () => { await writeFile(join(vault, "About Me", ".secret.md"), "# Hidden\n", "utf8") const outlines = await listMemoryFiles({ vaultPath: vault }, logger) diff --git a/src/vault-mcp/vault-operations/memory-store.ts b/src/vault-mcp/vault-operations/memory-store.ts index 2810ee5c2..6a6c2e34e 100644 --- a/src/vault-mcp/vault-operations/memory-store.ts +++ b/src/vault-mcp/vault-operations/memory-store.ts @@ -7,6 +7,7 @@ import { parseNote, stringifyNote } from "../obsidian-markdown/frontmatter.js" import { atomicWriteFile } from "./vault-filesystem.js" import { readFileOrNull, statOrNull } from "../../utils/fs.js" import { filterValidSymlinks } from "../../utils/filter-valid-symlinks.js" +import { mapWithConcurrency } from "../../utils/map-with-concurrency.js" import { isErrnoException } from "../../utils/is-errno-exception.js" import { describeError } from "../../utils/describe-error.js" import { assertNoControlCharacters } from "../../utils/assert-no-control-characters.js" @@ -677,12 +678,14 @@ export const createMemoryStore = (options: { memoryDir: string }) => { ): Promise => { if (!params.file) { const mdFiles = await listVisibleMemoryFilenames(params.vaultPath, logger) - const contents = await Promise.all( - mdFiles.map(async (filename) => { + const contents = await mapWithConcurrency({ + items: mdFiles, + concurrency: 16, + mapper: async (filename) => { const raw = await readListedMemoryFile({ vaultPath: params.vaultPath, filename }, logger) return parseNote(raw).content.trim() - }), - ) + }, + }) logger.info("get memory", { mode: "all", fileCount: mdFiles.length }) return contents.join("\n\n---\n\n") } @@ -949,8 +952,10 @@ export const createMemoryStore = (options: { memoryDir: string }) => { logger: Logger, ): Promise => { const mdFiles = await listVisibleMemoryFilenames(params.vaultPath, logger) - const outlines = await Promise.all( - mdFiles.map(async (filename) => { + const outlines = await mapWithConcurrency({ + items: mdFiles, + concurrency: 16, + mapper: async (filename) => { const raw = await readListedMemoryFile({ vaultPath: params.vaultPath, filename }, logger) const parsed = parseNote(raw) const name = basename(filename, ".md") @@ -978,8 +983,8 @@ export const createMemoryStore = (options: { memoryDir: string }) => { leading_callout: leadingCallout, headings, } - }), - ) + }, + }) logger.info("listed memory files", { count: outlines.length }) return outlines From 8fd8862f15b1ae2a4e921fdb00908a94a394d7ce Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:34:31 -0400 Subject: [PATCH 09/13] style(memory): explain the shared read limit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: triage · gpt-6-astra --- src/vault-mcp/vault-operations/memory-store.ts | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/src/vault-mcp/vault-operations/memory-store.ts b/src/vault-mcp/vault-operations/memory-store.ts index 6a6c2e34e..0c1afa6b1 100644 --- a/src/vault-mcp/vault-operations/memory-store.ts +++ b/src/vault-mcp/vault-operations/memory-store.ts @@ -347,6 +347,9 @@ const listSectionHeadings = (sections: readonly ParsedSection[]): string => { export const createMemoryStore = (options: { memoryDir: string }) => { const { memoryDir } = options + /** Large memory folders must not exhaust file handles by reading every file at once. */ + const memoryReadConcurrency = 16 + /** True for the .md entries the memory layer serves — excludes dot-prefixed * (hidden) filenames so a pre-existing hidden file on disk never leaks * through the no-file read or the list surfaces, mirroring the write-side @@ -680,7 +683,7 @@ export const createMemoryStore = (options: { memoryDir: string }) => { const mdFiles = await listVisibleMemoryFilenames(params.vaultPath, logger) const contents = await mapWithConcurrency({ items: mdFiles, - concurrency: 16, + concurrency: memoryReadConcurrency, mapper: async (filename) => { const raw = await readListedMemoryFile({ vaultPath: params.vaultPath, filename }, logger) return parseNote(raw).content.trim() @@ -954,7 +957,7 @@ export const createMemoryStore = (options: { memoryDir: string }) => { const mdFiles = await listVisibleMemoryFilenames(params.vaultPath, logger) const outlines = await mapWithConcurrency({ items: mdFiles, - concurrency: 16, + concurrency: memoryReadConcurrency, mapper: async (filename) => { const raw = await readListedMemoryFile({ vaultPath: params.vaultPath, filename }, logger) const parsed = parseNote(raw) From 911d64d9c5aae3d7292aade0c04845b8d199e483 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:38:02 -0400 Subject: [PATCH 10/13] fix: contain rebuild stat failures and resolve canvas links MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6.1-sol --- .../search/__tests__/file-watcher.test.ts | 72 ++++++ .../search/__tests__/search-index.test.ts | 225 +++++++++++++++++- src/vault-mcp/search/search-index.ts | 32 ++- 3 files changed, 322 insertions(+), 7 deletions(-) diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index 99b548929..fa4f4f463 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -523,6 +523,78 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { expect(errorSpy).toHaveBeenCalledTimes(1) }) + it.each([{ event: "add" }, { event: "change" }] as const)( + "contains a non-markdown read failure during $event and recovers", + async ({ event }) => { + const { testVault, search, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "content.txt") + await writeFile(filePath, "currentopal") + const actualFs = await vi.importActual("node:fs/promises") + const readSpy = vi.mocked(readFile).mockImplementation(async (requestedPath, options) => { + if (requestedPath === filePath) throw new Error("controlled content read failure") + return actualFs.readFile(requestedPath, options) + }) + const contentSpy = vi.spyOn(search, "upsertFileContent") + const warnSpy = vi.spyOn(logger, "warn") + onTestFinished(() => { + readSpy.mockRestore() + warnSpy.mockRestore() + }) + + await expect(fire(event, "content.txt")).resolves.toBeUndefined() + expect(readSpy).toHaveBeenCalledWith(filePath, "utf8") + expect(warnSpy).toHaveBeenCalledExactlyOnceWith("file content indexing failed", { + path: "content.txt", + error: "[Error]: controlled content read failure", + }) + expect(contentSpy).not.toHaveBeenCalled() + expect(database.prepare("SELECT path, bytes FROM non_md_files").all()).toEqual([ + { path: "content.txt", bytes: Buffer.byteLength("currentopal") }, + ]) + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([]) + readSpy.mockRestore() + await fire(event, "content.txt") + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ + { path: "content.txt", content: "currentopal" }, + ]) + expect(warnSpy).toHaveBeenCalledTimes(1) + }, + ) + + it.each([{ event: "add" }, { event: "change" }] as const)( + "contains a PDF extraction failure during $event and recovers", + async ({ event }) => { + const { testVault, search, database, fire } = await createControlledWatcher() + await writeFile(join(testVault, "doc.pdf"), "controlled PDF bytes") + const extractSpy = vi + .mocked(extractPdfText) + .mockRejectedValueOnce(new Error("controlled PDF extraction failure")) + .mockResolvedValueOnce({ text: "recoveredopal", totalPages: 1 }) + const contentSpy = vi.spyOn(search, "upsertFileContent") + const warnSpy = vi.spyOn(logger, "warn") + onTestFinished(() => { + extractSpy.mockRestore() + warnSpy.mockRestore() + }) + + await expect(fire(event, "doc.pdf")).resolves.toBeUndefined() + expect(extractSpy).toHaveBeenCalledTimes(1) + expect(warnSpy).toHaveBeenCalledExactlyOnceWith("file content indexing failed", { + path: "doc.pdf", + error: "[Error]: controlled PDF extraction failure", + }) + expect(contentSpy).not.toHaveBeenCalled() + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([{ path: "doc.pdf" }]) + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([]) + await fire(event, "doc.pdf") + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ + { path: "doc.pdf", content: "recoveredopal" }, + ]) + expect(extractSpy).toHaveBeenCalledTimes(2) + expect(warnSpy).toHaveBeenCalledTimes(1) + }, + ) + it("contains a file content removal failure during unlink and allows a later unlink", async () => { const { testVault, search, database, fire } = await createControlledWatcher() const filePath = join(testVault, "content.txt") diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index 165d9ba84..61f2261e5 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -1,5 +1,5 @@ import { describe, it, expect, vi, beforeEach, afterEach, onTestFinished } from "vitest" -import { mkdtemp, rm, writeFile, mkdir, symlink } from "node:fs/promises" +import { mkdtemp, rm, writeFile, mkdir, symlink, readFile } from "node:fs/promises" import { join } from "node:path" import { tmpdir } from "node:os" import Database from "better-sqlite3" @@ -10,6 +10,12 @@ import { createSearchIndex, INDEXABLE_TEXT_EXTENSIONS } from "../search-index.js import type { NoteMetadata, OutgoingLinkEntry, SearchIndex, TaskEntry } from "../search-index.js" import type { StatusClassification } from "../../obsidian-markdown/tasks.js" import { logger } from "../../../logger.js" +import { statOrNull } from "../../../utils/fs.js" +import { extractPdfText } from "../../obsidian-markdown/pdf.js" + +vi.mock("node:fs/promises", { spy: true }) +vi.mock("../../../utils/fs.js", { spy: true }) +vi.mock("../../obsidian-markdown/pdf.js", { spy: true }) const realSqliteVec = await vi.importActual("sqlite-vec") @@ -3256,6 +3262,141 @@ describe("markdown path requirement", () => { }) }) +describe("rebuildFromVault filesystem failures", () => { + const createRebuildVault = async () => { + const vaultPath = await mkdtemp(join(tmpdir(), "rebuild-error-")) + onTestFinished(() => rm(vaultPath, { recursive: true, force: true })) + const search = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true }) + await writeFile(join(vaultPath, "healthy.md"), "healthyamber") + return { vaultPath, search } + } + + it.each(["EACCES", "EIO"])("skips a non-markdown %s stat failure and recovers", async (code) => { + const { vaultPath, search } = await createRebuildVault() + const filePath = join(vaultPath, "image.png") + await writeFile(filePath, "image data") + const actualFs = + await vi.importActual("../../../utils/fs.js") + const statSpy = vi.mocked(statOrNull).mockImplementation(async (requestedPath) => { + if (requestedPath === filePath) + throw Object.assign(new Error("controlled stat failure"), { code }) + return actualFs.statOrNull(requestedPath) + }) + const warnSpy = vi.spyOn(logger, "warn") + onTestFinished(() => { + statSpy.mockRestore() + warnSpy.mockRestore() + }) + const rebuilt = await search.rebuildFromVault({ vaultPath }, logger) + await rebuilt.embedding + + expect(rebuilt.count).toBe(1) + expect(warnSpy).toHaveBeenCalledExactlyOnceWith("skipped unstattable file during rebuild", { + path: "image.png", + error: "[Error]: controlled stat failure", + }) + expect( + search.fullTextSearch({ query: "healthyamber" }, logger).map((entry) => entry.path), + ).toEqual(["healthy.md"]) + search.upsertNote( + { filePath: "source.md", rawContent: "![[image.png]]", fileStat: testStat(1000) }, + logger, + ) + expect( + search + .getOutgoingLinks({ path: "source.md" }, logger) + .map((link) => ({ path: link.path, exists: link.exists })), + ).toEqual([{ path: "image.png", exists: false }]) + statSpy.mockRestore() + const recovered = await search.rebuildFromVault({ vaultPath }, logger) + await recovered.embedding + search.upsertNote( + { filePath: "source.md", rawContent: "![[image.png]]", fileStat: testStat(1000) }, + logger, + ) + expect( + search + .getOutgoingLinks({ path: "source.md" }, logger) + .map((link) => ({ path: link.path, exists: link.exists })), + ).toEqual([{ path: "image.png", exists: true }]) + expect(warnSpy).toHaveBeenCalledTimes(1) + }) + + it.each([ + { fileName: "broken.md", sourceKind: "note" }, + { fileName: "broken.canvas", sourceKind: "canvas file" }, + { fileName: "broken.txt", sourceKind: "text file" }, + { fileName: "broken.pdf", sourceKind: "PDF" }, + ])( + "warns and skips an unreadable $sourceKind while indexing healthy files", + async ({ fileName, sourceKind }) => { + const { vaultPath, search } = await createRebuildVault() + const filePath = join(vaultPath, fileName) + await writeFile(filePath, "brokenquartz") + const actualFs = await vi.importActual("node:fs/promises") + const readSpy = vi.mocked(readFile).mockImplementation(async (requestedPath, options) => { + if (requestedPath === filePath) throw new Error("controlled read failure") + return actualFs.readFile(requestedPath, options) + }) + const warnSpy = vi.spyOn(logger, "warn") + onTestFinished(() => { + readSpy.mockRestore() + warnSpy.mockRestore() + }) + const rebuilt = await search.rebuildFromVault({ vaultPath }, logger) + await rebuilt.embedding + + expect(rebuilt.count).toBe(1) + expect(readSpy).toHaveBeenCalledWith(filePath, ...(fileName.endsWith(".pdf") ? [] : ["utf8"])) + expect(warnSpy).toHaveBeenCalledExactlyOnceWith( + `skipped unreadable ${sourceKind} during rebuild`, + { path: fileName, error: "[Error]: controlled read failure" }, + ) + expect( + (await search.hybridSearch({ query: "healthyamber" }, logger)).results.map( + (entry) => entry.path, + ), + ).toEqual(["healthy.md"]) + expect((await search.hybridSearch({ query: "brokenquartz" }, logger)).results).toEqual([]) + }, + ) + + it("contains a rebuild PDF extraction failure and indexes its next valid extraction", async () => { + const { vaultPath, search } = await createRebuildVault() + await writeFile(join(vaultPath, "broken.pdf"), "controlled bytes") + const extractSpy = vi + .mocked(extractPdfText) + .mockRejectedValueOnce(new Error("controlled PDF failure")) + .mockResolvedValueOnce({ text: "recoveredopal", totalPages: 1 }) + const warnSpy = vi.spyOn(logger, "warn") + onTestFinished(() => { + extractSpy.mockRestore() + warnSpy.mockRestore() + }) + const rebuilt = await search.rebuildFromVault({ vaultPath }, logger) + await rebuilt.embedding + expect(warnSpy).toHaveBeenCalledExactlyOnceWith("skipped unreadable PDF during rebuild", { + path: "broken.pdf", + error: "[Error]: controlled PDF failure", + }) + expect( + (await search.hybridSearch({ query: "healthyamber" }, logger)).results.map( + (entry) => entry.path, + ), + ).toEqual(["healthy.md"]) + expect((await search.hybridSearch({ query: "recoveredopal" }, logger)).results).toEqual([]) + const recovered = await search.rebuildFromVault({ vaultPath }, logger) + await recovered.embedding + expect( + (await search.hybridSearch({ query: "recoveredopal" }, logger)).results.map( + (entry) => entry.path, + ), + ).toEqual(["broken.pdf"]) + expect(extractSpy).toHaveBeenCalledTimes(2) + expect(warnSpy).toHaveBeenCalledTimes(1) + }) +}) + describe("rebuildFromVault", () => { let vaultDir: string @@ -5889,6 +6030,88 @@ describe("INDEXABLE_TEXT_EXTENSIONS", () => { // ── Canvas file content + link graph ────────────────────────── describe("canvas file content and links", () => { + it.each([{ fileToolsEnabled: true }, { fileToolsEnabled: false }])( + "resolves canvas links to existing notes and assets with file tools $fileToolsEnabled", + ({ fileToolsEnabled }) => { + const canvasIndex = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled }) + for (const notePath of ["Notes/Plan.md", "Notes/Route.md", "deep/Plan.md"]) { + canvasIndex.upsertNote( + { filePath: notePath, rawContent: "targetamber", fileStat: testStat(1000) }, + logger, + ) + } + for (const assetPath of [ + "photos/Sunset.png", + "photo.png.canvas", + "a/photo.png", + "Route.canvas", + "assets/map.canvas", + ]) { + canvasIndex.upsertNonMdFile(assetPath, 42) + } + canvasIndex.upsertNonMdFile("Boards/source.canvas", 100) + const targets = [ + "../Notes/Plan.md", + "Sunset.png", + "sunset.png", + "photo.png", + "Route", + "../assets/map.canvas", + "missing.png", + ] + const canvasContent = JSON.stringify({ + nodes: targets.map((file, position) => ({ + id: `file-${position}`, + type: "file", + x: 0, + y: position * 100, + width: 100, + height: 100, + file, + })), + edges: [], + }) + canvasIndex.upsertFileContent( + { + filePath: "Boards/source.canvas", + rawContent: canvasContent, + fileStat: testStat(1000, 100), + }, + logger, + ) + + expect( + canvasIndex + .getOutgoingLinks({ path: "Boards/source.canvas" }, logger) + .map((link) => ({ path: link.path, exists: link.exists })), + ).toEqual([ + { path: "Notes/Plan.md", exists: true }, + { path: "Notes/Route.md", exists: true }, + { path: "a/photo.png", exists: true }, + { path: "assets/map.canvas", exists: true }, + { path: "missing.png", exists: false }, + { path: "photos/Sunset.png", exists: true }, + ]) + expect( + canvasIndex.getBacklinks({ path: "Notes/Plan.md" }, logger).map((link) => link.path), + ).toEqual(["Boards/source.canvas"]) + expect( + canvasIndex.getBacklinks({ path: "assets/map.canvas" }, logger).map((link) => link.path), + ).toEqual(["Boards/source.canvas"]) + expect(canvasIndex.brokenLinkCount({}, logger)).toEqual({ + count: 1, + excludedFolder: null, + excludedCount: 0, + }) + canvasIndex.upsertNonMdFile("other/unchanged.txt", 12) + expect(canvasIndex.brokenLinkCount({}, logger)).toEqual({ + count: 1, + excludedFolder: null, + excludedCount: 0, + }) + }, + ) + const CANVAS_WITH_FILE_NODES = JSON.stringify({ nodes: [ { diff --git a/src/vault-mcp/search/search-index.ts b/src/vault-mcp/search/search-index.ts index 85d0927c8..fb9408005 100644 --- a/src/vault-mcp/search/search-index.ts +++ b/src/vault-mcp/search/search-index.ts @@ -1258,8 +1258,18 @@ export const createSearchIndex = ( // Link extraction — canvas only, unconditional (graph integrity) if (isCanvas) { deleteLinksStmt.run(params.filePath) - for (const linkTarget of canvasLinks) { - insertLinkStmt.run(params.filePath, linkTarget) + const notePaths = selectAllNotePathsStmt.all().map((note) => note.path) + for (const rawTarget of canvasLinks) { + const resolvedNotePath = links.resolve({ + target: rawTarget, + allPaths: notePaths, + sourcePath: params.filePath, + }) + const resolvedTarget = + resolvedNotePath ?? + resolveNonMarkdownFile({ target: rawTarget, sourcePath: params.filePath }) ?? + rawTarget + insertLinkStmt.run(params.filePath, resolvedTarget) } } })() @@ -2090,11 +2100,21 @@ export const createSearchIndex = ( // Stat non-md files before the write transaction (fs stays out of it). // A file vanishing between listing and stat (sync race) is dropped here // and re-indexed by its own watcher event. - const readFileSize = async (file: RebuildFilePaths) => { - const fileStat = await statOrNull(file.absolutePath) + const readFileSize = async ( + file: RebuildFilePaths, + ): Promise<{ relativePath: string; bytes: number } | null> => { + try { + const fileStat = await statOrNull(file.absolutePath) - if (!fileStat) return null - return { relativePath: file.relativePath, bytes: fileStat.size } + if (!fileStat) return null + return { relativePath: file.relativePath, bytes: fileStat.size } + } catch (error) { + logger.warn("skipped unstattable file during rebuild", { + path: file.relativePath, + error: describeError(error), + }) + return null + } } /** Bound filesystem work so a large vault cannot open every file at once. */ const REBUILD_IO_CONCURRENCY = 16 From 3ffe593af62977a428f086cbe18e5c5e715bcc7d Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:37:07 -0400 Subject: [PATCH 11/13] fix: treat vanished note sources as benign watcher races MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6.1-sol --- .../search/__tests__/file-watcher.test.ts | 93 +++++++++ .../search/__tests__/search-index.test.ts | 193 ++++++++++++++++++ src/vault-mcp/search/file-watcher.ts | 20 +- 3 files changed, 305 insertions(+), 1 deletion(-) diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index fa4f4f463..a0e9c7771 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -3,6 +3,7 @@ import { mkdtemp, rm, writeFile, + stat, mkdir, rename, unlink, @@ -658,6 +659,98 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { expect(errorSpy).toHaveBeenCalledTimes(1) }) + it.each([{ operation: "read" }, { operation: "stat" }] as const)( + "treats a note disappearing during its $operation as a benign source race", + async ({ operation }) => { + const { testVault, search, database, fire } = await createControlledWatcher() + const filePath = join(testVault, "note.md") + await writeFile(filePath, "committedamber") + await fire("add", "note.md") + const missingSource = Object.assign(new Error("controlled missing source"), { + code: "ENOENT", + }) + const sourceSpy = operation === "read" ? vi.mocked(readFile) : vi.mocked(stat) + sourceSpy.mockRejectedValueOnce(missingSource) + const upsertSpy = vi.spyOn(search, "upsertNote") + const embedSpy = vi.spyOn(search, "embedNote") + const debugSpy = vi.spyOn(logger, "debug") + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => { + sourceSpy.mockRestore() + debugSpy.mockRestore() + errorSpy.mockRestore() + }) + + await expect(fire("change", "note.md")).resolves.toBeUndefined() + + expect(debugSpy).toHaveBeenCalledExactlyOnceWith("change event skipped, file vanished", { + path: "note.md", + }) + expect(errorSpy).not.toHaveBeenCalled() + expect(upsertSpy).not.toHaveBeenCalled() + expect(embedSpy).not.toHaveBeenCalled() + expect(database.prepare("SELECT path, content FROM notes").all()).toEqual([ + { path: "note.md", content: "committedamber" }, + ]) + await unlink(filePath) + await fire("unlink", "note.md") + expect(database.prepare("SELECT path, content FROM notes").all()).toEqual([]) + await writeFile(filePath, "recoveredopal") + await fire("add", "note.md") + expect(database.prepare("SELECT path, content FROM notes").all()).toEqual([ + { path: "note.md", content: "recoveredopal" }, + ]) + expect(upsertSpy).toHaveBeenCalledTimes(1) + }, + ) + + it.each([ + { operation: "read", code: "EACCES" }, + { operation: "stat", code: "EIO" }, + { operation: "upsert", code: "ENOENT" }, + { operation: "embed", code: "ENOENT" }, + ] as const)("keeps $operation $code failures at error level", async ({ operation, code }) => { + const { testVault, search, fire } = await createControlledWatcher() + await writeFile(join(testVault, "note.md"), "currentamber") + const failure = Object.assign(new Error("controlled operation failure"), { code }) + const errorSpy = vi.spyOn(logger, "error") + const debugSpy = vi.spyOn(logger, "debug") + const restoreOperations: Array<() => void> = [] + onTestFinished(() => { + restoreOperations.forEach((restore) => restore()) + errorSpy.mockRestore() + debugSpy.mockRestore() + }) + if (operation === "read") { + vi.mocked(readFile).mockRejectedValueOnce(failure) + restoreOperations.push(() => vi.mocked(readFile).mockRestore()) + } + if (operation === "stat") { + vi.mocked(stat).mockRejectedValueOnce(failure) + restoreOperations.push(() => vi.mocked(stat).mockRestore()) + } + if (operation === "upsert") { + const upsertSpy = vi.spyOn(search, "upsertNote").mockImplementationOnce(() => { + throw failure + }) + restoreOperations.push(() => upsertSpy.mockRestore()) + } + if (operation === "embed") { + const embedSpy = vi.spyOn(search, "embedNote").mockRejectedValueOnce(failure) + restoreOperations.push(() => embedSpy.mockRestore()) + } + + await expect(fire("add", "note.md")).resolves.toBeUndefined() + + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to process file change", { + path: "note.md", + error: "[Error]: controlled operation failure", + }) + expect(debugSpy).not.toHaveBeenCalledWith("change event skipped, file vanished", { + path: "note.md", + }) + }) + const delayFirstRead = async (filePath: string) => { const actualFs = await vi.importActual("node:fs/promises") const entered = Promise.withResolvers() diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index 61f2261e5..d2b53b4fe 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -2,6 +2,7 @@ import { describe, it, expect, vi, beforeEach, afterEach, onTestFinished } from import { mkdtemp, rm, writeFile, mkdir, symlink, readFile } from "node:fs/promises" import { join } from "node:path" import { tmpdir } from "node:os" +import { setImmediate as setImmediateAsync } from "node:timers/promises" import Database from "better-sqlite3" import { DateTime } from "luxon" import * as sqliteVec from "sqlite-vec" @@ -3262,6 +3263,198 @@ describe("markdown path requirement", () => { }) }) +describe("rebuildFromVault bounded I/O", () => { + it.each([ + { operation: "size", extension: ".png", table: "non_md_files", decoyName: "healthy.md" }, + { operation: "note", extension: ".md", table: "notes", decoyName: "healthy.txt" }, + { operation: "text", extension: ".txt", table: "file_content", decoyName: "healthy.md" }, + { operation: "canvas", extension: ".canvas", table: "file_content", decoyName: "healthy.md" }, + ] as const)( + "bounds the $operation pass to 16 operations and indexes its seventeenth item", + async ({ operation, extension, table, decoyName }) => { + const directory = await mkdtemp(join(tmpdir(), "rebuild-bound-")) + const vaultPath = join(directory, "vault") + const releaseGate = Promise.withResolvers() + const boundReached = Promise.withResolvers() + const pendingRebuilds: Promise[] = [] + const restoreOperations: Array<() => void> = [] + const openDatabases: Database.Database[] = [] + onTestFinished(async () => { + releaseGate.resolve(undefined) + await Promise.allSettled(pendingRebuilds) + restoreOperations.forEach((restore) => restore()) + openDatabases.forEach((database) => database.close()) + await rm(directory, { recursive: true, force: true }) + }) + await mkdir(vaultPath) + const targetFiles = Array.from( + { length: 17 }, + (_unused, index) => `source-${String(index).padStart(2, "0")}${extension}`, + ) + for (const fileName of targetFiles) { + const content = + extension === ".canvas" + ? JSON.stringify({ + nodes: [ + { + id: "text", + type: "text", + x: 0, + y: 0, + width: 100, + height: 100, + text: "targetquartz", + }, + ], + edges: [], + }) + : "targetquartz" + await writeFile(join(vaultPath, fileName), content) + } + await writeFile(join(vaultPath, decoyName), "decoyamber") + const targetPaths = new Set(targetFiles.map((fileName) => join(vaultPath, fileName))) + const activePaths = new Set() + const startedPaths = new Set() + const activeCounts: number[] = [] + const holdTargetOperation = async (requestedPath: string): Promise => { + startedPaths.add(requestedPath) + activePaths.add(requestedPath) + activeCounts.push(activePaths.size) + if (activePaths.size === 16) boundReached.resolve(undefined) + await releaseGate.promise + activePaths.delete(requestedPath) + } + const actualFs = await vi.importActual("node:fs/promises") + const actualFsUtils = + await vi.importActual("../../../utils/fs.js") + + if (operation === "size") { + const statSpy = vi.mocked(statOrNull).mockImplementation(async (requestedPath) => { + if (targetPaths.has(requestedPath)) await holdTargetOperation(requestedPath) + return actualFsUtils.statOrNull(requestedPath) + }) + restoreOperations.push(() => statSpy.mockRestore()) + } + if (operation !== "size") { + const readSpy = vi.mocked(readFile).mockImplementation(async (requestedPath, options) => { + if (typeof requestedPath === "string" && targetPaths.has(requestedPath)) + await holdTargetOperation(requestedPath) + return actualFs.readFile(requestedPath, options) + }) + restoreOperations.push(() => readSpy.mockRestore()) + } + const dbPath = join(directory, "search.db") + const search = createSearchIndex(dbPath, undefined, undefined, { fileToolsEnabled: true }) + const database = new Database(dbPath, { readonly: true }) + openDatabases.push(database) + const rebuilding = search.rebuildFromVault({ vaultPath }, logger) + pendingRebuilds.push(rebuilding) + await boundReached.promise + await setImmediateAsync() + + expect(activePaths.size).toBe(16) + expect(startedPaths.size).toBe(16) + expect(database.prepare(`SELECT path FROM ${table}`).all()).toEqual([]) + releaseGate.resolve(undefined) + const rebuilt = await rebuilding + await rebuilt.embedding + + expect(Math.max(...activeCounts)).toBe(16) + expect(activePaths.size).toBe(0) + expect(startedPaths.size).toBe(17) + expect(database.prepare(`SELECT path FROM ${table} ORDER BY path`).all()).toEqual( + targetFiles.map((path) => ({ path })), + ) + expect( + (await search.hybridSearch({ query: "targetquartz" }, logger)).results + .map((entry) => entry.path) + .toSorted(), + ).toEqual(operation === "size" ? [] : targetFiles) + expect( + (await search.hybridSearch({ query: "decoyamber" }, logger)).results.map( + (entry) => entry.path, + ), + ).toEqual([decoyName]) + }, + ) + + it("bounds PDF extraction to four operations and indexes its fifth item", async () => { + const directory = await mkdtemp(join(tmpdir(), "rebuild-pdf-bound-")) + const vaultPath = join(directory, "vault") + const releaseGate = Promise.withResolvers() + const boundReached = Promise.withResolvers() + const pendingRebuilds: Promise[] = [] + const restoreOperations: Array<() => void> = [] + const openDatabases: Database.Database[] = [] + onTestFinished(async () => { + releaseGate.resolve(undefined) + await Promise.allSettled(pendingRebuilds) + restoreOperations.forEach((restore) => restore()) + openDatabases.forEach((database) => database.close()) + await rm(directory, { recursive: true, force: true }) + }) + await mkdir(vaultPath) + const targetFiles = Array.from({ length: 5 }, (_unused, index) => `source-${index}.pdf`) + const targetMarkers = new Set(targetFiles.map((fileName) => `marker-${fileName}`)) + for (const fileName of targetFiles) + await writeFile(join(vaultPath, fileName), `marker-${fileName}`) + await writeFile(join(vaultPath, "healthy.md"), "decoyamber") + await writeFile(join(vaultPath, "healthy.txt"), "decoyberyl") + const activeMarkers = new Set() + const startedMarkers = new Set() + const activeCounts: number[] = [] + const extractSpy = vi.mocked(extractPdfText).mockImplementation(async (pdfData) => { + const marker = Buffer.from(pdfData).toString("utf8") + + if (!targetMarkers.has(marker)) throw new Error(`unexpected PDF input: ${marker}`) + startedMarkers.add(marker) + activeMarkers.add(marker) + activeCounts.push(activeMarkers.size) + if (activeMarkers.size === 4) boundReached.resolve(undefined) + await releaseGate.promise + activeMarkers.delete(marker) + return { text: "pdfquartz", totalPages: 1 } + }) + restoreOperations.push(() => extractSpy.mockRestore()) + const dbPath = join(directory, "search.db") + const search = createSearchIndex(dbPath, undefined, undefined, { fileToolsEnabled: true }) + const database = new Database(dbPath, { readonly: true }) + openDatabases.push(database) + const rebuilding = search.rebuildFromVault({ vaultPath }, logger) + pendingRebuilds.push(rebuilding) + await boundReached.promise + await setImmediateAsync() + + expect(activeMarkers.size).toBe(4) + expect(startedMarkers.size).toBe(4) + releaseGate.resolve(undefined) + const rebuilt = await rebuilding + await rebuilt.embedding + + expect(Math.max(...activeCounts)).toBe(4) + expect(activeMarkers.size).toBe(0) + expect(startedMarkers.size).toBe(5) + expect(database.prepare("SELECT path FROM file_content ORDER BY path").all()).toEqual( + ["healthy.txt", ...targetFiles].map((path) => ({ path })), + ) + expect( + (await search.hybridSearch({ query: "pdfquartz" }, logger)).results + .map((entry) => entry.path) + .toSorted(), + ).toEqual(targetFiles) + expect( + (await search.hybridSearch({ query: "decoyamber" }, logger)).results.map( + (entry) => entry.path, + ), + ).toEqual(["healthy.md"]) + expect( + (await search.hybridSearch({ query: "decoyberyl" }, logger)).results.map( + (entry) => entry.path, + ), + ).toEqual(["healthy.txt"]) + }) +}) + describe("rebuildFromVault filesystem failures", () => { const createRebuildVault = async () => { const vaultPath = await mkdtemp(join(tmpdir(), "rebuild-error-")) diff --git a/src/vault-mcp/search/file-watcher.ts b/src/vault-mcp/search/file-watcher.ts index 9e617e94e..00db5cbe3 100644 --- a/src/vault-mcp/search/file-watcher.ts +++ b/src/vault-mcp/search/file-watcher.ts @@ -12,6 +12,7 @@ import type { SearchIndex } from "./search-index.js" import { extractPdfText } from "../obsidian-markdown/pdf.js" import { logger } from "../../logger.js" import { describeError } from "../../utils/describe-error.js" +import { isErrnoException } from "../../utils/is-errno-exception.js" import { readdirOrNull, realpathOrNull, statOrNull } from "../../utils/fs.js" import { hasHiddenPathSegment } from "../../utils/has-hidden-path-segment.js" @@ -170,9 +171,26 @@ export const startFileWatcher = ( } try { - const [content, fileStat] = await Promise.all([readFile(filePath, "utf8"), stat(filePath)]) + const readNoteSource = async (): Promise<{ content: string; fileStat: Stats } | null> => { + try { + const [content, fileStat] = await Promise.all([ + readFile(filePath, "utf8"), + stat(filePath), + ]) + return { content, fileStat } + } catch (error) { + if (!isErrnoException(error, "ENOENT")) throw error + + logger.debug("change event skipped, file vanished", { path: relativePath }) + return null + } + } + const noteSource = await readNoteSource() if (currentEvents.get(relativePath) !== eventToken) return + if (!noteSource) return + + const { content, fileStat } = noteSource const sourceVersion = search.upsertNote( { From 992d37066a172b58335295990f368297db31f9c8 Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:24:16 -0400 Subject: [PATCH 12/13] fix: complete independent cleanup and reuse canvas note catalogs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6.1-sol --- .../search/__tests__/file-watcher.test.ts | 142 ++++++++++++++- .../search/__tests__/search-index.test.ts | 171 ++++++++++++++++++ src/vault-mcp/search/file-watcher.ts | 3 +- src/vault-mcp/search/search-index.ts | 17 +- 4 files changed, 326 insertions(+), 7 deletions(-) diff --git a/src/vault-mcp/search/__tests__/file-watcher.test.ts b/src/vault-mcp/search/__tests__/file-watcher.test.ts index a0e9c7771..4eb65613f 100644 --- a/src/vault-mcp/search/__tests__/file-watcher.test.ts +++ b/src/vault-mcp/search/__tests__/file-watcher.test.ts @@ -488,7 +488,7 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { }, ) - it("contains an asset metadata removal failure during unlink and allows a later unlink", async () => { + it("removes independent file content after a metadata removal failure and retries metadata later", async () => { const { testVault, search, database, fire } = await createControlledWatcher() const filePath = join(testVault, "content.txt") await writeFile(filePath, "currentopal") @@ -505,7 +505,7 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { await expect(fire("unlink", "content.txt")).resolves.toBeUndefined() expect(removeSpy).toHaveBeenCalledExactlyOnceWith("content.txt") - expect(contentRemoveSpy).not.toHaveBeenCalled() + expect(contentRemoveSpy).toHaveBeenCalledExactlyOnceWith({ filePath: "content.txt" }, logger) expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to remove non-md file metadata", { path: "content.txt", error: "[Error]: controlled asset removal failure", @@ -514,14 +514,148 @@ describe("startFileWatcher — obsolete events and embedding queues", () => { expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([ { path: "content.txt" }, ]) + expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([]) + removeSpy.mockRestore() + await fire("unlink", "content.txt") + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) + expect(database.prepare("SELECT path FROM file_content").all()).toEqual([]) + expect(errorSpy).toHaveBeenCalledTimes(1) + }) + + it("reports both independent unlink failures and recovers on a later event", async () => { + const { testVault, search, database, fire } = await createControlledWatcher() + await writeFile(join(testVault, "content.txt"), "currentopal") + await fire("add", "content.txt") + await unlink(join(testVault, "content.txt")) + const metadataSpy = vi.spyOn(search, "removeNonMdFile").mockImplementationOnce(() => { + throw new Error("metadata removal failed") + }) + const contentSpy = vi.spyOn(search, "removeFileContent").mockImplementationOnce(() => { + throw new Error("content removal failed") + }) + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + + await fire("unlink", "content.txt") + + expect(metadataSpy).toHaveBeenCalledExactlyOnceWith("content.txt") + expect(contentSpy).toHaveBeenCalledExactlyOnceWith({ filePath: "content.txt" }, logger) + expect(errorSpy).toHaveBeenCalledTimes(2) + expect(errorSpy).toHaveBeenCalledWith("failed to remove non-md file metadata", { + path: "content.txt", + error: "[Error]: metadata removal failed", + }) + expect(errorSpy).toHaveBeenCalledWith("failed to remove file content", { + path: "content.txt", + error: "[Error]: content removal failed", + }) + expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([ + { path: "content.txt" }, + ]) expect(database.prepare("SELECT path, content FROM file_content").all()).toEqual([ { path: "content.txt", content: "currentopal" }, ]) - removeSpy.mockRestore() await fire("unlink", "content.txt") expect(database.prepare("SELECT path FROM non_md_files").all()).toEqual([]) expect(database.prepare("SELECT path FROM file_content").all()).toEqual([]) - expect(errorSpy).toHaveBeenCalledTimes(1) + expect(errorSpy).toHaveBeenCalledTimes(2) + }) + + it("cleans canvas links and vectors and rejects its held model after metadata removal fails", async () => { + const vector = new Float32Array(384).fill(0.1) + const modelEntered = Promise.withResolvers() + const releaseModel = Promise.withResolvers() + const modelFinished = Promise.withResolvers() + const embedder = { + embedText: vi + .fn() + .mockResolvedValueOnce(vector) + .mockImplementationOnce(() => { + modelEntered.resolve(undefined) + return releaseModel.promise + }), + embedBatch: vi.fn(async (texts: readonly string[]) => texts.map(() => vector)), + } + const { testVault, search, database, fire } = await createControlledWatcher(embedder) + const canvasContent = (text: string): string => + JSON.stringify({ + nodes: [ + { id: "text", type: "text", x: 0, y: 0, width: 100, height: 100, text }, + { id: "file", type: "file", x: 0, y: 200, width: 100, height: 100, file: "target.md" }, + ], + edges: [], + }) + search.upsertNote( + { filePath: "target.md", rawContent: "targetamber", fileStat: { mtimeMs: 1000, size: 11 } }, + logger, + ) + search.upsertNonMdFile("board.canvas", 100) + const initialVersion = search.upsertFileContent( + { + filePath: "board.canvas", + rawContent: canvasContent("oldquartz"), + fileStat: { mtimeMs: 1000, size: 100 }, + }, + logger, + ) + await search.embedFileContent( + { filePath: "board.canvas", sourceVersion: initialVersion }, + logger, + ) + expect(database.prepare("SELECT COUNT(*) AS count FROM file_content_vectors").get()).toEqual({ + count: 1, + }) + expect(search.getBacklinks({ path: "target.md" }, logger).map((link) => link.path)).toEqual([ + "board.canvas", + ]) + const realEmbed = search.embedFileContent + const pendingJobs: Promise[] = [] + vi.spyOn(search, "embedFileContent").mockImplementation((params, requestLogger) => { + const modelJob = realEmbed(params, requestLogger) + pendingJobs.push(modelJob) + // The detached job must settle before the fixture is removed. + return modelJob.finally(() => modelFinished.resolve(undefined)) + }) + onTestFinished(async () => { + releaseModel.resolve(vector) + await Promise.allSettled(pendingJobs) + }) + await writeFile(join(testVault, "board.canvas"), canvasContent("newopal")) + await fire("change", "board.canvas") + await modelEntered.promise + await unlink(join(testVault, "board.canvas")) + const metadataSpy = vi.spyOn(search, "removeNonMdFile").mockImplementationOnce(() => { + throw new Error("metadata removal failed") + }) + const errorSpy = vi.spyOn(logger, "error") + onTestFinished(() => errorSpy.mockRestore()) + + await fire("unlink", "board.canvas") + + expect(errorSpy).toHaveBeenCalledExactlyOnceWith("failed to remove non-md file metadata", { + path: "board.canvas", + error: "[Error]: metadata removal failed", + }) + expect( + database.prepare("SELECT path FROM non_md_files WHERE path = 'board.canvas'").all(), + ).toEqual([{ path: "board.canvas" }]) + expect(database.prepare("SELECT path FROM file_content").all()).toEqual([]) + expect(database.prepare("SELECT file_path FROM file_content_chunks").all()).toEqual([]) + expect(database.prepare("SELECT COUNT(*) AS count FROM file_content_vectors").get()).toEqual({ + count: 0, + }) + expect(search.getBacklinks({ path: "target.md" }, logger)).toEqual([]) + releaseModel.resolve(vector) + await modelFinished.promise + expect(database.prepare("SELECT file_path FROM file_content_chunks").all()).toEqual([]) + expect(database.prepare("SELECT COUNT(*) AS count FROM file_content_vectors").get()).toEqual({ + count: 0, + }) + await fire("unlink", "board.canvas") + expect( + database.prepare("SELECT path FROM non_md_files WHERE path = 'board.canvas'").all(), + ).toEqual([]) + expect(metadataSpy).toHaveBeenCalledTimes(2) }) it.each([{ event: "add" }, { event: "change" }] as const)( diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index d2b53b4fe..fa9ec3190 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -6222,6 +6222,177 @@ describe("INDEXABLE_TEXT_EXTENSIONS", () => { // ── Canvas file content + link graph ────────────────────────── +describe("canvas note catalog reuse", () => { + const observeCatalogScans = () => { + const scans = vi.fn() + const realPrepare: (this: Database.Database, source: string) => Database.Statement = + Database.prototype.prepare + /** The native method needs its SQLite receiver; count executions rather than preparations. */ + const prepareSpy = vi.spyOn(Database.prototype, "prepare").mockImplementation(function ( + this: Database.Database, + source: string, + ) { + const statement = realPrepare.call(this, source) + + if (source.trim() === "SELECT path FROM notes") { + const realAll = statement.all.bind(statement) + vi.spyOn(statement, "all").mockImplementation((...allParams: unknown[]) => { + scans() + return realAll(...allParams) + }) + } + return statement + }) + onTestFinished(() => prepareSpy.mockRestore()) + return scans + } + const canvasWithTarget = (target?: string): string => + JSON.stringify({ + nodes: target + ? [{ id: "file", type: "file", x: 0, y: 0, width: 100, height: 100, file: target }] + : [], + edges: [], + }) + const saveCanvas = (search: SearchIndex, target?: string): void => { + search.upsertFileContent( + { filePath: "board.canvas", rawContent: canvasWithTarget(target), fileStat: testStat(1000) }, + logger, + ) + } + const outgoingPaths = (search: SearchIndex): string[] => + search.getOutgoingLinks({ path: "board.canvas" }, logger).map((link) => link.path) + + it("avoids empty-canvas scans and shares one catalog across repeated linked saves", () => { + const scans = observeCatalogScans() + const search = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true }) + search.upsertNote( + { filePath: "Target.md", rawContent: "targetamber", fileStat: testStat(1000) }, + logger, + ) + search.upsertNote( + { filePath: "Decoy.md", rawContent: "decoyquartz", fileStat: testStat(1000) }, + logger, + ) + scans.mockClear() + saveCanvas(search) + expect(scans).not.toHaveBeenCalled() + expect(outgoingPaths(search)).toEqual([]) + for (const _unused of Array.from({ length: 12 })) saveCanvas(search, "Target") + expect(scans).toHaveBeenCalledTimes(1) + expect(outgoingPaths(search)).toEqual(["Target.md"]) + saveCanvas(search) + expect(outgoingPaths(search)).toEqual([]) + expect(scans).toHaveBeenCalledTimes(1) + }) + + it("refreshes the catalog after note additions, deletions and recreation with asset fallback", () => { + const scans = observeCatalogScans() + const search = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true }) + search.upsertNonMdFile("Target.canvas", 100) + saveCanvas(search, "Target") + expect(outgoingPaths(search)).toEqual(["Target.canvas"]) + search.upsertNote( + { filePath: "Target.md", rawContent: "targetamber", fileStat: testStat(1000) }, + logger, + ) + scans.mockClear() + saveCanvas(search, "Target") + expect(scans).toHaveBeenCalledTimes(1) + expect(outgoingPaths(search)).toEqual(["Target.md"]) + search.removeNote("Target.md") + scans.mockClear() + saveCanvas(search, "Target") + expect(scans).toHaveBeenCalledTimes(1) + expect(outgoingPaths(search)).toEqual(["Target.canvas"]) + search.upsertNote( + { filePath: "Target.md", rawContent: "recreatedopal", fileStat: testStat(2000) }, + logger, + ) + scans.mockClear() + saveCanvas(search, "Target") + saveCanvas(search, "Target") + expect(scans).toHaveBeenCalledTimes(1) + expect(outgoingPaths(search)).toEqual(["Target.md"]) + }) + + it("keeps committed membership after a failed note upsert rolls back", () => { + const poison = installStatementPoison("INSERT INTO tasks") + const search = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true }) + search.upsertNonMdFile("Target.canvas", 100) + saveCanvas(search, "Target") + poison.arm() + expect(() => + search.upsertNote( + { + filePath: "Target.md", + rawContent: "- [ ] task that triggers poison", + fileStat: testStat(1000), + }, + logger, + ), + ).toThrow(poison.message) + poison.disarm() + saveCanvas(search, "Target") + expect(outgoingPaths(search)).toEqual(["Target.canvas"]) + }) + + it("uses the rebuild catalog for a canvas corpus and later saves", async () => { + const scans = observeCatalogScans() + const directory = await mkdtemp(join(tmpdir(), "canvas-catalog-")) + onTestFinished(() => rm(directory, { recursive: true, force: true })) + await writeFile(join(directory, "Target.md"), "targetamber") + await writeFile(join(directory, "Decoy.md"), "decoyquartz") + const canvasNames = Array.from({ length: 9 }, (_unused, position) => `board-${position}.canvas`) + for (const canvasName of canvasNames) + await writeFile(join(directory, canvasName), canvasWithTarget("Target")) + const search = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true }) + const rebuilt = await search.rebuildFromVault({ vaultPath: directory }, logger) + await rebuilt.embedding + expect(rebuilt.count).toBe(2) + expect(scans).toHaveBeenCalledTimes(1) + expect(search.getBacklinks({ path: "Target.md" }, logger).map((link) => link.path)).toEqual( + canvasNames, + ) + saveCanvas(search, "Target") + expect(scans).toHaveBeenCalledTimes(1) + expect(outgoingPaths(search)).toEqual(["Target.md"]) + }) + + it("drops an uncommitted rebuild catalog on a late outer rollback", async () => { + const scans = observeCatalogScans() + const directory = await mkdtemp(join(tmpdir(), "canvas-catalog-rollback-")) + onTestFinished(() => rm(directory, { recursive: true, force: true })) + await writeFile(join(directory, "Target.md"), "targetamber") + await writeFile(join(directory, "failed.canvas"), canvasWithTarget("Target")) + const search = createSearchIndex(":memory:", undefined, undefined, { fileToolsEnabled: true }) + search.upsertNote( + { filePath: "Existing.md", rawContent: "existingquartz", fileStat: testStat(1000) }, + logger, + ) + saveCanvas(search, "Existing") + const failure = new Error("controlled late rebuild rollback") + const rebuildLogger = logger.child({ operation: "controlled-rebuild" }) + vi.spyOn(rebuildLogger, "debug").mockImplementation((message) => { + if (message === "indexed file content") throw failure + }) + vi.spyOn(rebuildLogger, "warn").mockImplementation(() => { + throw failure + }) + await expect(search.rebuildFromVault({ vaultPath: directory }, rebuildLogger)).rejects.toThrow( + failure.message, + ) + search.upsertNonMdFile("Target.canvas", 100) + scans.mockClear() + saveCanvas(search, "Target") + expect(scans).toHaveBeenCalledTimes(1) + expect(outgoingPaths(search)).toEqual(["Target.canvas"]) + const recovered = await search.rebuildFromVault({ vaultPath: directory }, logger) + await recovered.embedding + saveCanvas(search, "Target") + expect(outgoingPaths(search)).toEqual(["Target.md"]) + }) +}) + describe("canvas file content and links", () => { it.each([{ fileToolsEnabled: true }, { fileToolsEnabled: false }])( "resolves canvas links to existing notes and assets with file tools $fileToolsEnabled", diff --git a/src/vault-mcp/search/file-watcher.ts b/src/vault-mcp/search/file-watcher.ts index 00db5cbe3..a43cba80f 100644 --- a/src/vault-mcp/search/file-watcher.ts +++ b/src/vault-mcp/search/file-watcher.ts @@ -254,7 +254,6 @@ export const startFileWatcher = ( path: relativePath, error: describeError(error), }) - return } const deletedExtension = extname(filePath) const isDeletedCanvas = deletedExtension === ".canvas" @@ -272,7 +271,7 @@ export const startFileWatcher = ( return } } - logger.debug("removed non-md file from index", { path: relativePath }) + logger.debug("processed non-md file removal", { path: relativePath }) return } diff --git a/src/vault-mcp/search/search-index.ts b/src/vault-mcp/search/search-index.ts index fb9408005..a37279fee 100644 --- a/src/vault-mcp/search/search-index.ts +++ b/src/vault-mcp/search/search-index.ts @@ -653,6 +653,16 @@ export const createSearchIndex = ( ) const selectAllNotePathsStmt = db.prepare(`SELECT path FROM notes`) + /** Canvas saves share this catalog until a committed note write or rebuild changes it. */ + let cachedCanvasNotePaths: readonly string[] | undefined + const getCanvasNotePaths = (): readonly string[] => { + if (cachedCanvasNotePaths) return cachedCanvasNotePaths + + const notePaths = selectAllNotePathsStmt.all().map((note) => note.path) + cachedCanvasNotePaths = notePaths + return notePaths + } + // ── Non-markdown file awareness ──────────────────────────────── // // Obsidian resolves extensionless wikilinks (e.g. [[Trip Route]]) against @@ -1258,7 +1268,7 @@ export const createSearchIndex = ( // Link extraction — canvas only, unconditional (graph integrity) if (isCanvas) { deleteLinksStmt.run(params.filePath) - const notePaths = selectAllNotePathsStmt.all().map((note) => note.path) + const notePaths = canvasLinks.length > 0 ? getCanvasNotePaths() : [] for (const rawTarget of canvasLinks) { const resolvedNotePath = links.resolve({ target: rawTarget, @@ -1605,6 +1615,7 @@ export const createSearchIndex = ( } })() + cachedCanvasNotePaths = undefined const sourceVersion = Symbol() sourceVersions.set(filePath, sourceVersion) @@ -2014,6 +2025,7 @@ export const createSearchIndex = ( removeMemoryEntriesForFile(memoryFile) } })() + cachedCanvasNotePaths = undefined sourceVersions.delete(filePath) } @@ -2028,6 +2040,7 @@ export const createSearchIndex = ( const orphanFileVectorsRemoved = deleteOrphanFileVectorsStmt?.run().changes ?? 0 const orphanMemoryVectorsRemoved = deleteOrphanMemoryVectorsStmt?.run().changes ?? 0 sourceVersions.clear() + cachedCanvasNotePaths = undefined db.exec("DELETE FROM notes_fts") db.exec("DELETE FROM notes") db.exec("DELETE FROM links") @@ -2305,6 +2318,7 @@ export const createSearchIndex = ( // (e.g. Note A links to Note B, but Note B was indexed after Note A). const allPaths = selectAllNotePathsStmt.all() const pathList = allPaths.map((row) => row.path) + cachedCanvasNotePaths = pathList db.exec("DELETE FROM links") for (const note of noteContents) { @@ -2366,6 +2380,7 @@ export const createSearchIndex = ( })() } catch (error) { sourceVersions.clear() + cachedCanvasNotePaths = undefined throw error } From 5439fd09af778568929ee4ac398f5f7f8fb4821b Mon Sep 17 00:00:00 2001 From: Tanisha Aberdeen <32620895+aliasunder@users.noreply.github.com> Date: Tue, 6 Oct 2026 00:28:03 -0400 Subject: [PATCH 13/13] test: assert exact catalog rollback errors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ship-Check: pr-monitor · gpt-6.1-sol --- src/vault-mcp/search/__tests__/search-index.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/vault-mcp/search/__tests__/search-index.test.ts b/src/vault-mcp/search/__tests__/search-index.test.ts index fa9ec3190..95ca9f0fd 100644 --- a/src/vault-mcp/search/__tests__/search-index.test.ts +++ b/src/vault-mcp/search/__tests__/search-index.test.ts @@ -6330,7 +6330,7 @@ describe("canvas note catalog reuse", () => { }, logger, ), - ).toThrow(poison.message) + ).toThrow(new Error(poison.message)) poison.disarm() saveCanvas(search, "Target") expect(outgoingPaths(search)).toEqual(["Target.canvas"]) @@ -6379,7 +6379,7 @@ describe("canvas note catalog reuse", () => { throw failure }) await expect(search.rebuildFromVault({ vaultPath: directory }, rebuildLogger)).rejects.toThrow( - failure.message, + failure, ) search.upsertNonMdFile("Target.canvas", 100) scans.mockClear()