From 029dd5a11e91bc871f332ccf5f1c0d3933e73a89 Mon Sep 17 00:00:00 2001 From: David Wong Date: Thu, 24 Sep 2026 10:40:24 +0300 Subject: [PATCH] =?UTF-8?q?feat(admission):=20=E5=86=85=E5=AE=B9=E5=93=88?= =?UTF-8?q?=E5=B8=8C=E8=AE=A1=E9=87=8F=E4=B8=8E=E5=BD=92=E6=A1=A3=E8=A1=8C?= =?UTF-8?q?=E7=BA=B3=E5=85=A5=E5=8E=BB=E9=87=8D=E5=80=99=E9=80=89=E9=9B=86?= =?UTF-8?q?=EF=BC=88#254=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 维护者 2026-09-24 拍板的两条实现侧信号,仍只计量不拦截: 1. 计量信号用内容哈希,不用相似度:新增 content_hash 派生列(归一化后 sha256), 幂等迁移、索引、存量回填,五个写入口全部重算;写入准入的 metadata.dup 记下 命中行的 id 与归档状态。 2. 归档行纳入去重候选集:findContentHashMatches 的候选集含归档行(scope 三维按 IS 比较)。归档出口止体积不止重复,剩下的穿透口靠它兜住。 归一化口径(折叠了什么、为什么命中不等于内容相同)写在 src/content-hash.js 文件头; 索引列序 hash 打头由 EXPLAIN QUERY PLAN 用例锁住,否则存量回填每次开库退化成全表扫。 --- dsh-mneme/lib/content-hash.js | 54 +++++++ dsh-mneme/lib/store.js | 110 +++++++++++-- dsh-mneme/lib/write-admission.js | 66 +++++++- dsh-mneme/src/content-hash.js | 54 +++++++ dsh-mneme/src/store.js | 110 +++++++++++-- dsh-mneme/src/write-admission.js | 66 +++++++- dsh-mneme/test/content-hash.test.js | 204 +++++++++++++++++++++++++ dsh-mneme/test/write-admission.test.js | 87 +++++++++++ 8 files changed, 713 insertions(+), 38 deletions(-) create mode 100644 dsh-mneme/lib/content-hash.js create mode 100644 dsh-mneme/src/content-hash.js create mode 100644 dsh-mneme/test/content-hash.test.js diff --git a/dsh-mneme/lib/content-hash.js b/dsh-mneme/lib/content-hash.js new file mode 100644 index 00000000..d72fe9d3 --- /dev/null +++ b/dsh-mneme/lib/content-hash.js @@ -0,0 +1,54 @@ +// #254 写入准入的确定性计量锚:内容归一化哈希(exact duplicate 判定)。 +// +// 为什么用哈希而不是相似度:计量阶段要的是一个零成本、可复算、不留争议的判据。 +// 相似度阈值两边都是误判——同一件事换个说法就掉到线下,不同的事共享话题词又顶到 +// 线上,阈值往哪边挪都只是换一种错法(#254 拍板:计量信号用内容哈希)。哈希只认 +// 「归一化后逐字节相同」,算不出来就只有一种解释。 +// +// 但它不等于「原文逐字节相同」:归一化会折掉格式,也会折掉标点,于是 `版本 3.5` 与 +// `版本 35`、`a-b` 与 `ab` 落同一个键。折叠范围(大小写 / 空白 / 标点)是口径本身 +// 定的,#254 已拍板——误报面在计量阶段可接受(只标记、不拦截),但「命中即精确重复」 +// 这句话别升级成「命中即内容相同」。要更窄的判据就得改归一化口径,而那要带存量重算。 +// +// 归一化口径一次定死(NFKC → 小写 → 去标点 → 空白折叠 → trim),动机是让「只差 +// 格式」的两次写入落在同一个键上:全角/半角、大小写、行内换行与缩进、句末标点都 +// 不该让同一件事变成两条。反过来它不折叠同义词、不排序词序——那些只能靠相似度, +// 而相似度已判定不可用于本信号。 +// +// 改口径必须带存量重算策略:memories.content_hash 是派生列,口径一变,库里已有的 +// 值全部按旧口径算——补 NULL 修不好,只能整列重算(UPDATE 全表 + 重建索引)。 +// 所以口径只能在这里改,且改要单独成批,不要混进别的改动里。 +// +// 空内容(标题与正文归一后都为空)返回 null:这类写入没有可比内容,落 NULL 让它们 +// 不互相匹配成「全是重复」。 +// +// 另有一处形状相近、口径不同的哈希:service.js 里镜像人工编辑的 digest 校验(两处) +// 用 sha256(title\0content),那里**不**归一化(要逐字节保真)。两把哈希用途不同,别互换。 +import { createHash } from "node:crypto"; + +/** + * 归一化一段文本用于内容比对(口径见文件头)。 + * @param {unknown} value + * @returns {string} + */ +export function normalizeForHash(value) { + return String(value ?? "") + .normalize("NFKC") + .toLowerCase() + .replace(/\p{P}/gu, "") + .replace(/\s+/g, " ") + .trim(); +} + +/** + * 一行记忆的内容锚:标题与正文分别归一后用 NUL 拼接再取 sha256(十六进制全串)。 + * NUL 分隔是防拼接歧义——('ab','c') 与 ('a','bc') 不能落到同一个键上。 + * @param {{title?: string, content?: string}|null|undefined} memory + * @returns {string|null} 归一后无内容时返回 null + */ +export function contentHashOf(memory) { + const title = normalizeForHash(memory?.title); + const content = normalizeForHash(memory?.content); + if (!title && !content) return null; + return createHash("sha256").update(`${title}\u0000${content}`).digest("hex"); +} diff --git a/dsh-mneme/lib/store.js b/dsh-mneme/lib/store.js index 879c2997..fc514490 100644 --- a/dsh-mneme/lib/store.js +++ b/dsh-mneme/lib/store.js @@ -1,6 +1,7 @@ import { DatabaseSync } from "node:sqlite"; import { randomUUID } from "node:crypto"; import { statSync } from "node:fs"; +import { contentHashOf } from "./content-hash.js"; const SCHEMA = ` CREATE TABLE IF NOT EXISTS memories ( @@ -14,6 +15,7 @@ CREATE TABLE IF NOT EXISTS memories ( archived INTEGER NOT NULL DEFAULT 0, source TEXT, content_history TEXT, + content_hash TEXT, embedding TEXT, epistemic_status TEXT NOT NULL DEFAULT 'subjective', last_accessed_at TEXT, @@ -407,6 +409,7 @@ function toRow(row) { type: row.type, title: row.title, content: row.content, + content_hash: row.content_hash ?? undefined, tags: parseTags(row.tags), importance: row.importance, forgotten: row.forgotten === 1, @@ -715,6 +718,34 @@ export function createStore(path) { addColumn("memories", "workspace_scope_source", "ALTER TABLE memories ADD COLUMN workspace_scope_source TEXT"); addColumn("memories", "scope_decided_at", "ALTER TABLE memories ADD COLUMN scope_decided_at TEXT"); + // #254 计量信号:内容归一化哈希(口径与动机写在 content-hash.js 的文件头)。派生 + // 列,不参与任何判定——只给「同一内容又被写了一次」当等值锚。索引建在加列之后: + // 老库打开时 SCHEMA 的 CREATE TABLE 对既有表不生效,列还不存在,把索引写进 SCHEMA + // 会直接报 no such column(与下方 llm_audit_logs.session_key 同理)。 + addColumn("memories", "content_hash", "ALTER TABLE memories ADD COLUMN content_hash TEXT"); + // 列序是 content_hash 打头:哈希几乎唯一,等值 seek 就已收窄到候选行,type 只当同 + // 一次 seek 里的第二列过滤;反过来(type 打头)下面的存量回填就得全表扫——每次打开 + // 都要把全部正文读一遍。 + db.exec("CREATE INDEX IF NOT EXISTS idx_memories_content_hash ON memories(content_hash, type)"); + // 存量回填:老库(含归档区)的行都没有值,而「归档行纳入去重候选集」这条口径正是 + // 要靠既有归档行出数据。只算 NULL 行、幂等;改口径不是补 NULL 能修好的,要整列重算 + // (见 content-hash.js 的文件头)。空标题空正文的行算不出锚,跳过不写——否则每次 + // 打开都要为这些行再跑一遍 UPDATE。 + { + const updates = db + .prepare("SELECT id, title, content FROM memories WHERE content_hash IS NULL") + .all() + .map((row) => [contentHashOf(row), row.id]) + .filter(([hash]) => hash); + if (updates.length) { + const stmt = db.prepare("UPDATE memories SET content_hash = ? WHERE id = ? AND content_hash IS NULL"); + // 一个事务包住整批:逐条自动提交是每行一次 WAL 提交,几千行的存量库上纯属浪费。 + runAtomically(() => { + for (const [hash, id] of updates) stmt.run(hash, id); + }); + } + } + // Legacy dream_runs without policy_epoch → backfill with the default epoch. addColumn("dream_runs", "policy_epoch", "ALTER TABLE dream_runs ADD COLUMN policy_epoch INTEGER NOT NULL DEFAULT 0"); addColumn("dream_runs", "run_type", "ALTER TABLE dream_runs ADD COLUMN run_type TEXT NOT NULL DEFAULT 'auto'"); @@ -953,6 +984,8 @@ export function createStore(path) { const importance = Number.isInteger(memory.importance) ? memory.importance : 3; const evidence = Array.isArray(memory.evidence) ? JSON.stringify(memory.evidence) : null; const docPath = typeof memory.doc_path === "string" && memory.doc_path.trim() ? memory.doc_path : null; + // #254 计量锚:由本行的 title/content 派生,调用方传什么都以这里算出的为准。 + const contentHash = contentHashOf(memory); const embedding = Array.isArray(memory.embedding) && memory.embedding.length ? JSON.stringify(memory.embedding) : null; @@ -963,8 +996,8 @@ export function createStore(path) { : inferEpistemicStatus(memory); runAtomically(() => { db.prepare( - `INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, content_history, quality_score, embedding, epistemic_status, agent_scope, workspace_scope, agent_scope_source, workspace_scope_source, scope_decided_at, sensitivity, occurred_at, evidence, doc_path, created_at, updated_at) - VALUES (?, ?, ?, ?, ?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + `INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, content_history, quality_score, embedding, epistemic_status, agent_scope, workspace_scope, agent_scope_source, workspace_scope_source, scope_decided_at, sensitivity, occurred_at, evidence, doc_path, content_hash, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` ).run( id, type, @@ -987,6 +1020,7 @@ export function createStore(path) { normalizeOccurredAt(memory.occurred_at), evidence, docPath, + contentHash, now, now ); @@ -999,6 +1033,41 @@ export function createStore(path) { return getById(id); } + /** + * #254 计量:与本次写入「归一化内容哈希」相同的既有行(去重候选集)。 + * 结构化锚先收窄成本:按哈希等值 seek(走 idx_memories_content_hash,哈希几乎唯 + * 一,落在候选集上的行只有几条),再按 type 与 scope 三维过滤——不是全表扫、也不 + * 逐行比哈希。scope 用 IS 比(NULL 安全):未标注行互相匹配、与已标注行不匹配, + * 与 service.js 去重键(scopeKeyOf)同口径;入参先按 store 自己的 normalizeScopeText + * 归一,免得调用方传 " foo " 就漏配。 + * 归档行与已遗忘行都在集内:#275 的分界是「出口止体积、不止重复」——被质量闸归档 + * 的同一个事实必须仍能判成重复,否则同样的内容再写一次又是一条新行(#254 拍板: + * 归档行纳入去重候选集)。 + * 排序即取舍:活跃行排在归档/遗忘行之前,再按 updated_at 倒序。LIMIT 先于调用方的 + * 「活区优先」判断执行,而归档动作本身会顶 updated_at——纯按时间倒序时,同键命中一 + * 旦超过窗口宽度,活跃行就被归档行挤出候选集,调用方只能看到归档命中,把「活跃重复」 + * 误报成「归档重复」,恰好把这条信号要分流的两类弄反。 + * @returns {Array<{id: string, archived: boolean, forgotten: boolean}>} 活跃行优先,其后按最近写入 + */ + function findContentHashMatches({ type, hash, agent_scope: agentScope, workspace_scope: workspaceScope, sensitivity, limit = 10 } = {}) { + if (!hash || !type) return []; + const lim = Number.isInteger(limit) && limit > 0 ? Math.min(limit, 50) : 10; + const rows = db.prepare( + `SELECT id, archived, forgotten FROM memories + WHERE type = ? AND content_hash = ? + AND agent_scope IS ? AND workspace_scope IS ? AND sensitivity IS ? + ORDER BY archived ASC, forgotten ASC, updated_at DESC, id LIMIT ?` + ).all( + type, + hash, + normalizeScopeText(agentScope), + normalizeScopeText(workspaceScope), + normalizeScopeText(sensitivity), + lim + ); + return rows.map((row) => ({ id: row.id, archived: row.archived === 1, forgotten: row.forgotten === 1 })); + } + function update(id, patch) { const existing = getById(id); if (!existing) throw new Error(`memory not found: ${id}`); @@ -1045,13 +1114,18 @@ export function createStore(path) { const nextDecidedAt = patch.scope_decided_at !== undefined ? normalizeOccurredAt(patch.scope_decided_at) : (existing.scope_decided_at ?? null); + // #254:标题或正文变了,锚跟着重算——它是当前内容的派生物,留旧值等于让「这行 + // 现在装的什么」和标记对不上。 + const nextTitle = patch.title ?? existing.title; + const nextContent = patch.content ?? existing.content; + const contentHash = contentHashOf({ title: nextTitle, content: nextContent }); runAtomically(() => { db.prepare( - `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, agent_scope=?, workspace_scope=?, agent_scope_source=?, workspace_scope_source=?, scope_decided_at=?, evidence=?, updated_at=? WHERE id=?` + `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, agent_scope=?, workspace_scope=?, agent_scope_source=?, workspace_scope_source=?, scope_decided_at=?, evidence=?, content_hash=?, updated_at=? WHERE id=?` ).run( type, - patch.title ?? existing.title, - patch.content ?? existing.content, + nextTitle, + nextContent, JSON.stringify(patch.tags ?? existing.tags), Number.isInteger(patch.importance) ? patch.importance : existing.importance, patch.source !== undefined ? patch.source : (existing.source ?? null), @@ -1065,6 +1139,7 @@ export function createStore(path) { nextWorkspaceSource, nextDecidedAt, evidence, + contentHash, now, id ); @@ -1126,15 +1201,19 @@ export function createStore(path) { // dirty == false — recoverMirror sees no debt and the mirror stays stale. // Wrapping both in one transaction means a CAS miss rolls back cleanly too // (no write, no generation bump). + // 同 update:#254 的锚随标题/正文重算(CAS 路径也改这两列)。 + const nextTitle = patch.title ?? existing.title; + const nextContent = patch.content ?? existing.content; + const contentHash = contentHashOf({ title: nextTitle, content: nextContent }); let applied = false; runAtomically(() => { const result = db.prepare( - `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, updated_at=? + `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, content_hash=?, updated_at=? WHERE id=? AND updated_at=?` ).run( type, - patch.title ?? existing.title, - patch.content ?? existing.content, + nextTitle, + nextContent, JSON.stringify(patch.tags ?? existing.tags), Number.isInteger(patch.importance) ? patch.importance : existing.importance, patch.source !== undefined ? patch.source : (existing.source ?? null), @@ -1142,6 +1221,7 @@ export function createStore(path) { qualityScore, embedding, epistemicStatus, + contentHash, now, id, expectedUpdatedAt @@ -1193,15 +1273,15 @@ export function createStore(path) { function demoteToSummary(id, summary, { minRefTimeMs } = {}) { let changed = false; runAtomically(() => { - const row = db.prepare("SELECT last_accessed_at, content, _full_content FROM memories WHERE id = ?").get(id); + const row = db.prepare("SELECT title, last_accessed_at, content, _full_content FROM memories WHERE id = ?").get(id); if (!row || row._full_content) return; if (minRefTimeMs !== undefined && row.last_accessed_at) { const lastMs = Date.parse(row.last_accessed_at); if (lastMs >= minRefTimeMs) return; // touched after snapshot — still hot } db.prepare( - "UPDATE memories SET content = ?, _full_content = ?, updated_at = ? WHERE id = ?" - ).run(summary, row.content, nowIso(), id); + "UPDATE memories SET content = ?, _full_content = ?, content_hash = ?, updated_at = ? WHERE id = ?" + ).run(summary, row.content, contentHashOf({ title: row.title, content: summary }), nowIso(), id); incrementGeneration(); changed = true; }); @@ -1212,11 +1292,11 @@ export function createStore(path) { function restoreContent(id) { let changed = false; runAtomically(() => { - const row = db.prepare("SELECT content, _full_content FROM memories WHERE id = ?").get(id); + const row = db.prepare("SELECT title, content, _full_content FROM memories WHERE id = ?").get(id); if (!row || !row._full_content) return; db.prepare( - "UPDATE memories SET content = ?, _full_content = NULL, updated_at = ? WHERE id = ?" - ).run(row._full_content, nowIso(), id); + "UPDATE memories SET content = ?, _full_content = NULL, content_hash = ?, updated_at = ? WHERE id = ?" + ).run(row._full_content, contentHashOf({ title: row.title, content: row._full_content }), nowIso(), id); incrementGeneration(); changed = true; }); @@ -2606,6 +2686,8 @@ export function createStore(path) { saveDocument, update, compareAndUpdate, + // #254:写入准入的内容哈希候选集(只读)。 + findContentHashMatches, remove, setForget, setArchived, diff --git a/dsh-mneme/lib/write-admission.js b/dsh-mneme/lib/write-admission.js index 6bf8581c..2899a1ad 100644 --- a/dsh-mneme/lib/write-admission.js +++ b/dsh-mneme/lib/write-admission.js @@ -21,10 +21,22 @@ // 观察穿透频率。穿透行既不参与 g2 判定,也不推进同话题的时间基准——它整个不在闸门 // 里,成为下一次比较的基准会把 g2 的样本混进穿透流量。 // +// 内容哈希(exact duplicate,第三类测量点,维护者 2026-09-24 拍板):归一化口径定在 +// content-hash.js,命中即「同一内容又被写了一次」。判据用哈希而不是相似度——相似度 +// 阈值在同一件事换个说法 / 不同的事共享话题词之间来回挪,只会换一种错法。候选集先按 +// type 与 scope 三维等值收窄再比哈希(不是全表扫),并把归档行一并纳入:「出口止体积、 +// 不止重复」,被质量闸归档的同一个事实不在 saveWithDedupe 的候选集里(store.list 默认 +// 排除归档),同样的内容再写一次就是一条新行,这正是这条信号要量的穿透。两个穿透口与 +// pinned 豁免同口径:pinned 在这里返回得更早,连候选集查询都不发。 +// +// 信号跟着同一行审计走,所以内容哈希与 g1/g2 覆盖同一批写入(会话内新建行);无会话 +// 身份的系统写入(dream / summarize / import / organize)要另立行形状,不在本批。 +// // 与 llmAudit.enabled 的关系:那个开关同时关掉审计行的启动期清理(index.js 的 // deleteOldLlmAudits),所以关掉时本模块一行都不写,否则就是在无保留期的表里做 // 按写入频次增长。 import { PINNED_MEMORY_TYPES } from "./service.js"; +import { contentHashOf } from "./content-hash.js"; // 审计口径(llm_audit_logs 的既有列):trigger_source = 组件名,operation_type = // 动作名,status='skipped' = 本行没有产生任何 LLM 花费(与 summarize 的间隔门、 @@ -123,17 +135,48 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = } /** - * 阶段一决策:恒 allow(本模块里没有阈值),只算出两个闸门的测量点。返回形状一次 + * 内容哈希候选集查询(#254 第三类测量点)。失败降级为「没命中」:漏一个测量点, + * 不影响写入,也不让计量反噬。 + * 多条命中时的取舍(活区优先):有活跃/未遗忘的命中就报它(判据是「去重候选集本该 + * 拦住」),只有归档/遗忘行命中才报它们。store 侧已把活跃行排在窗口头部,这里的 find + * 是防御性重复:排序口径将来再变,本判据也不跟着变。 + * archived 与 forgotten 分开回:两者都是「出口」,但穿透含义不同——只命中已遗忘行时 + * 若只回 archived=false,读审计的人会以为库里有活跃重复;而 saveWithDedupe 的候选集 + * 本来就排除遗忘行(store.list 默认 includeForgotten=false),这类命中同样是穿透。 + * @returns {{memory_id: string, archived: boolean, forgotten: boolean}|null} + */ + function lookupContentDup(memory) { + try { + const hash = contentHashOf(memory); + if (!hash) return null; + const hits = store?.findContentHashMatches?.({ + type: memory?.type, + hash, + agent_scope: memory?.agent_scope, + workspace_scope: memory?.workspace_scope, + sensitivity: memory?.sensitivity + }) ?? []; + const hit = hits.find((h) => !h.archived && !h.forgotten) ?? hits[0]; + return hit ? { memory_id: hit.id, archived: hit.archived === true, forgotten: hit.forgotten === true } : null; + } catch (e) { + warn(`[dsh-mneme] write admission content hash lookup failed: ${String(e)}`); + return null; + } + } + + /** + * 阶段一决策:恒 allow(本模块里没有阈值),只算出三个闸门的测量点。返回形状一次 * 定死,阶段二只往里加分支、不改字段: * decision — "allow" | "confirm"(阶段一只可能 allow) * gate — "g1" | null(null = 这次写入不进预算,不记行) * topics — 本次写入的确定性话题锚 * repeat — {topic, gapMs} | null(g2 的命中面) + * dup — {memory_id, archived, forgotten} | null(内容哈希的命中面) * exempt — null | "pinned" * @param {{memory: object, sessionKey?: string|null}} input */ function evaluate({ memory, sessionKey } = {}) { - const verdict = { decision: "allow", gate: null, topics: [], repeat: null, exempt: null }; + const verdict = { decision: "allow", gate: null, topics: [], repeat: null, dup: null, exempt: null }; // 无会话身份 = 系统写入(dream / summarize / import / organize),不进预算。 if (!sessionKey) return verdict; // 审计关掉时不记行、也不推进话题表(见文件头:那个开关连启动期清理一起关掉)。 @@ -141,6 +184,7 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = const topics = extractTopicKeys(memory); const pinned = PINNED_MEMORY_TYPES.has(String(memory?.type ?? "")); if (pinned) return { ...verdict, gate: ADMISSION_GATE_SESSION_BUDGET, topics, exempt: "pinned" }; + const dup = lookupContentDup(memory); const table = topicTable(sessionKey); const at = now(); let repeat = null; @@ -150,16 +194,19 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = // 同一行可能命中多个锚:取最近的那次(间隔最短=最有价值的那次重复)。 if (!repeat || gapMs < repeat.gapMs) repeat = { topic, gapMs }; } - return { ...verdict, gate: ADMISSION_GATE_SESSION_BUDGET, topics, repeat }; + return { ...verdict, gate: ADMISSION_GATE_SESSION_BUDGET, topics, repeat, dup }; } /** - * 把测量点落成审计行(一行一个新建行)。两个信号的读法: + * 把测量点落成审计行(一行一个新建行)。三个信号的读法: * g1 会话写入预算:`GROUP BY session_key` 计数即得「会话内新建条数」分布;要剔 * 掉穿透行(pinned 不进预算)就加 `json_extract(metadata,'$.exempt') IS NULL`。 * g2 同话题冷却:`json_extract(metadata,'$.g2.gap_ms')` 即得「同话题新建行间隔」 * 分布。间隔算的是同一话题两次**新建行**之间——标题命中走并入的那次不在这里 * (并入正是冷却要做的事,已经做到了)。 + * dup 内容哈希:`json_extract(metadata,'$.dup.memory_id')` 非空即「这条新行的内容 + * 与某条既有行归一化后逐字节相同」;`$.dup.archived` 区分命中那条在活区还是 + * 归档区——归档区命中就是去重候选集漏掉的那一类穿透。 * 计量绝不影响写入:任何异常只 warn。 * @returns {object[]} 落下的审计行(无测量点或失败时为空数组) */ @@ -183,7 +230,16 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = gate: verdict.gate, topics, ...(verdict.exempt ? { exempt: verdict.exempt } : {}), - ...(verdict.repeat ? { g2: { topic: verdict.repeat.topic, gap_ms: verdict.repeat.gapMs } } : {}) + ...(verdict.repeat ? { g2: { topic: verdict.repeat.topic, gap_ms: verdict.repeat.gapMs } } : {}), + // dup 两个出口分字段落盘:archived 与 forgotten 是两类不同的穿透,合成一个布尔 + // 会让「只剩遗忘行命中」读起来像「库里有活跃重复」(见 lookupContentDup 注释)。 + ...(verdict.dup + ? { dup: { + memory_id: verdict.dup.memory_id, + archived: verdict.dup.archived === true, + forgotten: verdict.dup.forgotten === true + } } + : {}) } })); } catch (e) { diff --git a/dsh-mneme/src/content-hash.js b/dsh-mneme/src/content-hash.js new file mode 100644 index 00000000..d72fe9d3 --- /dev/null +++ b/dsh-mneme/src/content-hash.js @@ -0,0 +1,54 @@ +// #254 写入准入的确定性计量锚:内容归一化哈希(exact duplicate 判定)。 +// +// 为什么用哈希而不是相似度:计量阶段要的是一个零成本、可复算、不留争议的判据。 +// 相似度阈值两边都是误判——同一件事换个说法就掉到线下,不同的事共享话题词又顶到 +// 线上,阈值往哪边挪都只是换一种错法(#254 拍板:计量信号用内容哈希)。哈希只认 +// 「归一化后逐字节相同」,算不出来就只有一种解释。 +// +// 但它不等于「原文逐字节相同」:归一化会折掉格式,也会折掉标点,于是 `版本 3.5` 与 +// `版本 35`、`a-b` 与 `ab` 落同一个键。折叠范围(大小写 / 空白 / 标点)是口径本身 +// 定的,#254 已拍板——误报面在计量阶段可接受(只标记、不拦截),但「命中即精确重复」 +// 这句话别升级成「命中即内容相同」。要更窄的判据就得改归一化口径,而那要带存量重算。 +// +// 归一化口径一次定死(NFKC → 小写 → 去标点 → 空白折叠 → trim),动机是让「只差 +// 格式」的两次写入落在同一个键上:全角/半角、大小写、行内换行与缩进、句末标点都 +// 不该让同一件事变成两条。反过来它不折叠同义词、不排序词序——那些只能靠相似度, +// 而相似度已判定不可用于本信号。 +// +// 改口径必须带存量重算策略:memories.content_hash 是派生列,口径一变,库里已有的 +// 值全部按旧口径算——补 NULL 修不好,只能整列重算(UPDATE 全表 + 重建索引)。 +// 所以口径只能在这里改,且改要单独成批,不要混进别的改动里。 +// +// 空内容(标题与正文归一后都为空)返回 null:这类写入没有可比内容,落 NULL 让它们 +// 不互相匹配成「全是重复」。 +// +// 另有一处形状相近、口径不同的哈希:service.js 里镜像人工编辑的 digest 校验(两处) +// 用 sha256(title\0content),那里**不**归一化(要逐字节保真)。两把哈希用途不同,别互换。 +import { createHash } from "node:crypto"; + +/** + * 归一化一段文本用于内容比对(口径见文件头)。 + * @param {unknown} value + * @returns {string} + */ +export function normalizeForHash(value) { + return String(value ?? "") + .normalize("NFKC") + .toLowerCase() + .replace(/\p{P}/gu, "") + .replace(/\s+/g, " ") + .trim(); +} + +/** + * 一行记忆的内容锚:标题与正文分别归一后用 NUL 拼接再取 sha256(十六进制全串)。 + * NUL 分隔是防拼接歧义——('ab','c') 与 ('a','bc') 不能落到同一个键上。 + * @param {{title?: string, content?: string}|null|undefined} memory + * @returns {string|null} 归一后无内容时返回 null + */ +export function contentHashOf(memory) { + const title = normalizeForHash(memory?.title); + const content = normalizeForHash(memory?.content); + if (!title && !content) return null; + return createHash("sha256").update(`${title}\u0000${content}`).digest("hex"); +} diff --git a/dsh-mneme/src/store.js b/dsh-mneme/src/store.js index 879c2997..fc514490 100644 --- a/dsh-mneme/src/store.js +++ b/dsh-mneme/src/store.js @@ -1,6 +1,7 @@ import { DatabaseSync } from "node:sqlite"; import { randomUUID } from "node:crypto"; import { statSync } from "node:fs"; +import { contentHashOf } from "./content-hash.js"; const SCHEMA = ` CREATE TABLE IF NOT EXISTS memories ( @@ -14,6 +15,7 @@ CREATE TABLE IF NOT EXISTS memories ( archived INTEGER NOT NULL DEFAULT 0, source TEXT, content_history TEXT, + content_hash TEXT, embedding TEXT, epistemic_status TEXT NOT NULL DEFAULT 'subjective', last_accessed_at TEXT, @@ -407,6 +409,7 @@ function toRow(row) { type: row.type, title: row.title, content: row.content, + content_hash: row.content_hash ?? undefined, tags: parseTags(row.tags), importance: row.importance, forgotten: row.forgotten === 1, @@ -715,6 +718,34 @@ export function createStore(path) { addColumn("memories", "workspace_scope_source", "ALTER TABLE memories ADD COLUMN workspace_scope_source TEXT"); addColumn("memories", "scope_decided_at", "ALTER TABLE memories ADD COLUMN scope_decided_at TEXT"); + // #254 计量信号:内容归一化哈希(口径与动机写在 content-hash.js 的文件头)。派生 + // 列,不参与任何判定——只给「同一内容又被写了一次」当等值锚。索引建在加列之后: + // 老库打开时 SCHEMA 的 CREATE TABLE 对既有表不生效,列还不存在,把索引写进 SCHEMA + // 会直接报 no such column(与下方 llm_audit_logs.session_key 同理)。 + addColumn("memories", "content_hash", "ALTER TABLE memories ADD COLUMN content_hash TEXT"); + // 列序是 content_hash 打头:哈希几乎唯一,等值 seek 就已收窄到候选行,type 只当同 + // 一次 seek 里的第二列过滤;反过来(type 打头)下面的存量回填就得全表扫——每次打开 + // 都要把全部正文读一遍。 + db.exec("CREATE INDEX IF NOT EXISTS idx_memories_content_hash ON memories(content_hash, type)"); + // 存量回填:老库(含归档区)的行都没有值,而「归档行纳入去重候选集」这条口径正是 + // 要靠既有归档行出数据。只算 NULL 行、幂等;改口径不是补 NULL 能修好的,要整列重算 + // (见 content-hash.js 的文件头)。空标题空正文的行算不出锚,跳过不写——否则每次 + // 打开都要为这些行再跑一遍 UPDATE。 + { + const updates = db + .prepare("SELECT id, title, content FROM memories WHERE content_hash IS NULL") + .all() + .map((row) => [contentHashOf(row), row.id]) + .filter(([hash]) => hash); + if (updates.length) { + const stmt = db.prepare("UPDATE memories SET content_hash = ? WHERE id = ? AND content_hash IS NULL"); + // 一个事务包住整批:逐条自动提交是每行一次 WAL 提交,几千行的存量库上纯属浪费。 + runAtomically(() => { + for (const [hash, id] of updates) stmt.run(hash, id); + }); + } + } + // Legacy dream_runs without policy_epoch → backfill with the default epoch. addColumn("dream_runs", "policy_epoch", "ALTER TABLE dream_runs ADD COLUMN policy_epoch INTEGER NOT NULL DEFAULT 0"); addColumn("dream_runs", "run_type", "ALTER TABLE dream_runs ADD COLUMN run_type TEXT NOT NULL DEFAULT 'auto'"); @@ -953,6 +984,8 @@ export function createStore(path) { const importance = Number.isInteger(memory.importance) ? memory.importance : 3; const evidence = Array.isArray(memory.evidence) ? JSON.stringify(memory.evidence) : null; const docPath = typeof memory.doc_path === "string" && memory.doc_path.trim() ? memory.doc_path : null; + // #254 计量锚:由本行的 title/content 派生,调用方传什么都以这里算出的为准。 + const contentHash = contentHashOf(memory); const embedding = Array.isArray(memory.embedding) && memory.embedding.length ? JSON.stringify(memory.embedding) : null; @@ -963,8 +996,8 @@ export function createStore(path) { : inferEpistemicStatus(memory); runAtomically(() => { db.prepare( - `INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, content_history, quality_score, embedding, epistemic_status, agent_scope, workspace_scope, agent_scope_source, workspace_scope_source, scope_decided_at, sensitivity, occurred_at, evidence, doc_path, created_at, updated_at) - VALUES (?, ?, ?, ?, ?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + `INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, content_history, quality_score, embedding, epistemic_status, agent_scope, workspace_scope, agent_scope_source, workspace_scope_source, scope_decided_at, sensitivity, occurred_at, evidence, doc_path, content_hash, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` ).run( id, type, @@ -987,6 +1020,7 @@ export function createStore(path) { normalizeOccurredAt(memory.occurred_at), evidence, docPath, + contentHash, now, now ); @@ -999,6 +1033,41 @@ export function createStore(path) { return getById(id); } + /** + * #254 计量:与本次写入「归一化内容哈希」相同的既有行(去重候选集)。 + * 结构化锚先收窄成本:按哈希等值 seek(走 idx_memories_content_hash,哈希几乎唯 + * 一,落在候选集上的行只有几条),再按 type 与 scope 三维过滤——不是全表扫、也不 + * 逐行比哈希。scope 用 IS 比(NULL 安全):未标注行互相匹配、与已标注行不匹配, + * 与 service.js 去重键(scopeKeyOf)同口径;入参先按 store 自己的 normalizeScopeText + * 归一,免得调用方传 " foo " 就漏配。 + * 归档行与已遗忘行都在集内:#275 的分界是「出口止体积、不止重复」——被质量闸归档 + * 的同一个事实必须仍能判成重复,否则同样的内容再写一次又是一条新行(#254 拍板: + * 归档行纳入去重候选集)。 + * 排序即取舍:活跃行排在归档/遗忘行之前,再按 updated_at 倒序。LIMIT 先于调用方的 + * 「活区优先」判断执行,而归档动作本身会顶 updated_at——纯按时间倒序时,同键命中一 + * 旦超过窗口宽度,活跃行就被归档行挤出候选集,调用方只能看到归档命中,把「活跃重复」 + * 误报成「归档重复」,恰好把这条信号要分流的两类弄反。 + * @returns {Array<{id: string, archived: boolean, forgotten: boolean}>} 活跃行优先,其后按最近写入 + */ + function findContentHashMatches({ type, hash, agent_scope: agentScope, workspace_scope: workspaceScope, sensitivity, limit = 10 } = {}) { + if (!hash || !type) return []; + const lim = Number.isInteger(limit) && limit > 0 ? Math.min(limit, 50) : 10; + const rows = db.prepare( + `SELECT id, archived, forgotten FROM memories + WHERE type = ? AND content_hash = ? + AND agent_scope IS ? AND workspace_scope IS ? AND sensitivity IS ? + ORDER BY archived ASC, forgotten ASC, updated_at DESC, id LIMIT ?` + ).all( + type, + hash, + normalizeScopeText(agentScope), + normalizeScopeText(workspaceScope), + normalizeScopeText(sensitivity), + lim + ); + return rows.map((row) => ({ id: row.id, archived: row.archived === 1, forgotten: row.forgotten === 1 })); + } + function update(id, patch) { const existing = getById(id); if (!existing) throw new Error(`memory not found: ${id}`); @@ -1045,13 +1114,18 @@ export function createStore(path) { const nextDecidedAt = patch.scope_decided_at !== undefined ? normalizeOccurredAt(patch.scope_decided_at) : (existing.scope_decided_at ?? null); + // #254:标题或正文变了,锚跟着重算——它是当前内容的派生物,留旧值等于让「这行 + // 现在装的什么」和标记对不上。 + const nextTitle = patch.title ?? existing.title; + const nextContent = patch.content ?? existing.content; + const contentHash = contentHashOf({ title: nextTitle, content: nextContent }); runAtomically(() => { db.prepare( - `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, agent_scope=?, workspace_scope=?, agent_scope_source=?, workspace_scope_source=?, scope_decided_at=?, evidence=?, updated_at=? WHERE id=?` + `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, agent_scope=?, workspace_scope=?, agent_scope_source=?, workspace_scope_source=?, scope_decided_at=?, evidence=?, content_hash=?, updated_at=? WHERE id=?` ).run( type, - patch.title ?? existing.title, - patch.content ?? existing.content, + nextTitle, + nextContent, JSON.stringify(patch.tags ?? existing.tags), Number.isInteger(patch.importance) ? patch.importance : existing.importance, patch.source !== undefined ? patch.source : (existing.source ?? null), @@ -1065,6 +1139,7 @@ export function createStore(path) { nextWorkspaceSource, nextDecidedAt, evidence, + contentHash, now, id ); @@ -1126,15 +1201,19 @@ export function createStore(path) { // dirty == false — recoverMirror sees no debt and the mirror stays stale. // Wrapping both in one transaction means a CAS miss rolls back cleanly too // (no write, no generation bump). + // 同 update:#254 的锚随标题/正文重算(CAS 路径也改这两列)。 + const nextTitle = patch.title ?? existing.title; + const nextContent = patch.content ?? existing.content; + const contentHash = contentHashOf({ title: nextTitle, content: nextContent }); let applied = false; runAtomically(() => { const result = db.prepare( - `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, updated_at=? + `UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, content_hash=?, updated_at=? WHERE id=? AND updated_at=?` ).run( type, - patch.title ?? existing.title, - patch.content ?? existing.content, + nextTitle, + nextContent, JSON.stringify(patch.tags ?? existing.tags), Number.isInteger(patch.importance) ? patch.importance : existing.importance, patch.source !== undefined ? patch.source : (existing.source ?? null), @@ -1142,6 +1221,7 @@ export function createStore(path) { qualityScore, embedding, epistemicStatus, + contentHash, now, id, expectedUpdatedAt @@ -1193,15 +1273,15 @@ export function createStore(path) { function demoteToSummary(id, summary, { minRefTimeMs } = {}) { let changed = false; runAtomically(() => { - const row = db.prepare("SELECT last_accessed_at, content, _full_content FROM memories WHERE id = ?").get(id); + const row = db.prepare("SELECT title, last_accessed_at, content, _full_content FROM memories WHERE id = ?").get(id); if (!row || row._full_content) return; if (minRefTimeMs !== undefined && row.last_accessed_at) { const lastMs = Date.parse(row.last_accessed_at); if (lastMs >= minRefTimeMs) return; // touched after snapshot — still hot } db.prepare( - "UPDATE memories SET content = ?, _full_content = ?, updated_at = ? WHERE id = ?" - ).run(summary, row.content, nowIso(), id); + "UPDATE memories SET content = ?, _full_content = ?, content_hash = ?, updated_at = ? WHERE id = ?" + ).run(summary, row.content, contentHashOf({ title: row.title, content: summary }), nowIso(), id); incrementGeneration(); changed = true; }); @@ -1212,11 +1292,11 @@ export function createStore(path) { function restoreContent(id) { let changed = false; runAtomically(() => { - const row = db.prepare("SELECT content, _full_content FROM memories WHERE id = ?").get(id); + const row = db.prepare("SELECT title, content, _full_content FROM memories WHERE id = ?").get(id); if (!row || !row._full_content) return; db.prepare( - "UPDATE memories SET content = ?, _full_content = NULL, updated_at = ? WHERE id = ?" - ).run(row._full_content, nowIso(), id); + "UPDATE memories SET content = ?, _full_content = NULL, content_hash = ?, updated_at = ? WHERE id = ?" + ).run(row._full_content, contentHashOf({ title: row.title, content: row._full_content }), nowIso(), id); incrementGeneration(); changed = true; }); @@ -2606,6 +2686,8 @@ export function createStore(path) { saveDocument, update, compareAndUpdate, + // #254:写入准入的内容哈希候选集(只读)。 + findContentHashMatches, remove, setForget, setArchived, diff --git a/dsh-mneme/src/write-admission.js b/dsh-mneme/src/write-admission.js index 6bf8581c..2899a1ad 100644 --- a/dsh-mneme/src/write-admission.js +++ b/dsh-mneme/src/write-admission.js @@ -21,10 +21,22 @@ // 观察穿透频率。穿透行既不参与 g2 判定,也不推进同话题的时间基准——它整个不在闸门 // 里,成为下一次比较的基准会把 g2 的样本混进穿透流量。 // +// 内容哈希(exact duplicate,第三类测量点,维护者 2026-09-24 拍板):归一化口径定在 +// content-hash.js,命中即「同一内容又被写了一次」。判据用哈希而不是相似度——相似度 +// 阈值在同一件事换个说法 / 不同的事共享话题词之间来回挪,只会换一种错法。候选集先按 +// type 与 scope 三维等值收窄再比哈希(不是全表扫),并把归档行一并纳入:「出口止体积、 +// 不止重复」,被质量闸归档的同一个事实不在 saveWithDedupe 的候选集里(store.list 默认 +// 排除归档),同样的内容再写一次就是一条新行,这正是这条信号要量的穿透。两个穿透口与 +// pinned 豁免同口径:pinned 在这里返回得更早,连候选集查询都不发。 +// +// 信号跟着同一行审计走,所以内容哈希与 g1/g2 覆盖同一批写入(会话内新建行);无会话 +// 身份的系统写入(dream / summarize / import / organize)要另立行形状,不在本批。 +// // 与 llmAudit.enabled 的关系:那个开关同时关掉审计行的启动期清理(index.js 的 // deleteOldLlmAudits),所以关掉时本模块一行都不写,否则就是在无保留期的表里做 // 按写入频次增长。 import { PINNED_MEMORY_TYPES } from "./service.js"; +import { contentHashOf } from "./content-hash.js"; // 审计口径(llm_audit_logs 的既有列):trigger_source = 组件名,operation_type = // 动作名,status='skipped' = 本行没有产生任何 LLM 花费(与 summarize 的间隔门、 @@ -123,17 +135,48 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = } /** - * 阶段一决策:恒 allow(本模块里没有阈值),只算出两个闸门的测量点。返回形状一次 + * 内容哈希候选集查询(#254 第三类测量点)。失败降级为「没命中」:漏一个测量点, + * 不影响写入,也不让计量反噬。 + * 多条命中时的取舍(活区优先):有活跃/未遗忘的命中就报它(判据是「去重候选集本该 + * 拦住」),只有归档/遗忘行命中才报它们。store 侧已把活跃行排在窗口头部,这里的 find + * 是防御性重复:排序口径将来再变,本判据也不跟着变。 + * archived 与 forgotten 分开回:两者都是「出口」,但穿透含义不同——只命中已遗忘行时 + * 若只回 archived=false,读审计的人会以为库里有活跃重复;而 saveWithDedupe 的候选集 + * 本来就排除遗忘行(store.list 默认 includeForgotten=false),这类命中同样是穿透。 + * @returns {{memory_id: string, archived: boolean, forgotten: boolean}|null} + */ + function lookupContentDup(memory) { + try { + const hash = contentHashOf(memory); + if (!hash) return null; + const hits = store?.findContentHashMatches?.({ + type: memory?.type, + hash, + agent_scope: memory?.agent_scope, + workspace_scope: memory?.workspace_scope, + sensitivity: memory?.sensitivity + }) ?? []; + const hit = hits.find((h) => !h.archived && !h.forgotten) ?? hits[0]; + return hit ? { memory_id: hit.id, archived: hit.archived === true, forgotten: hit.forgotten === true } : null; + } catch (e) { + warn(`[dsh-mneme] write admission content hash lookup failed: ${String(e)}`); + return null; + } + } + + /** + * 阶段一决策:恒 allow(本模块里没有阈值),只算出三个闸门的测量点。返回形状一次 * 定死,阶段二只往里加分支、不改字段: * decision — "allow" | "confirm"(阶段一只可能 allow) * gate — "g1" | null(null = 这次写入不进预算,不记行) * topics — 本次写入的确定性话题锚 * repeat — {topic, gapMs} | null(g2 的命中面) + * dup — {memory_id, archived, forgotten} | null(内容哈希的命中面) * exempt — null | "pinned" * @param {{memory: object, sessionKey?: string|null}} input */ function evaluate({ memory, sessionKey } = {}) { - const verdict = { decision: "allow", gate: null, topics: [], repeat: null, exempt: null }; + const verdict = { decision: "allow", gate: null, topics: [], repeat: null, dup: null, exempt: null }; // 无会话身份 = 系统写入(dream / summarize / import / organize),不进预算。 if (!sessionKey) return verdict; // 审计关掉时不记行、也不推进话题表(见文件头:那个开关连启动期清理一起关掉)。 @@ -141,6 +184,7 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = const topics = extractTopicKeys(memory); const pinned = PINNED_MEMORY_TYPES.has(String(memory?.type ?? "")); if (pinned) return { ...verdict, gate: ADMISSION_GATE_SESSION_BUDGET, topics, exempt: "pinned" }; + const dup = lookupContentDup(memory); const table = topicTable(sessionKey); const at = now(); let repeat = null; @@ -150,16 +194,19 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = // 同一行可能命中多个锚:取最近的那次(间隔最短=最有价值的那次重复)。 if (!repeat || gapMs < repeat.gapMs) repeat = { topic, gapMs }; } - return { ...verdict, gate: ADMISSION_GATE_SESSION_BUDGET, topics, repeat }; + return { ...verdict, gate: ADMISSION_GATE_SESSION_BUDGET, topics, repeat, dup }; } /** - * 把测量点落成审计行(一行一个新建行)。两个信号的读法: + * 把测量点落成审计行(一行一个新建行)。三个信号的读法: * g1 会话写入预算:`GROUP BY session_key` 计数即得「会话内新建条数」分布;要剔 * 掉穿透行(pinned 不进预算)就加 `json_extract(metadata,'$.exempt') IS NULL`。 * g2 同话题冷却:`json_extract(metadata,'$.g2.gap_ms')` 即得「同话题新建行间隔」 * 分布。间隔算的是同一话题两次**新建行**之间——标题命中走并入的那次不在这里 * (并入正是冷却要做的事,已经做到了)。 + * dup 内容哈希:`json_extract(metadata,'$.dup.memory_id')` 非空即「这条新行的内容 + * 与某条既有行归一化后逐字节相同」;`$.dup.archived` 区分命中那条在活区还是 + * 归档区——归档区命中就是去重候选集漏掉的那一类穿透。 * 计量绝不影响写入:任何异常只 warn。 * @returns {object[]} 落下的审计行(无测量点或失败时为空数组) */ @@ -183,7 +230,16 @@ export function createWriteAdmission({ store, config, logger, now = Date.now } = gate: verdict.gate, topics, ...(verdict.exempt ? { exempt: verdict.exempt } : {}), - ...(verdict.repeat ? { g2: { topic: verdict.repeat.topic, gap_ms: verdict.repeat.gapMs } } : {}) + ...(verdict.repeat ? { g2: { topic: verdict.repeat.topic, gap_ms: verdict.repeat.gapMs } } : {}), + // dup 两个出口分字段落盘:archived 与 forgotten 是两类不同的穿透,合成一个布尔 + // 会让「只剩遗忘行命中」读起来像「库里有活跃重复」(见 lookupContentDup 注释)。 + ...(verdict.dup + ? { dup: { + memory_id: verdict.dup.memory_id, + archived: verdict.dup.archived === true, + forgotten: verdict.dup.forgotten === true + } } + : {}) } })); } catch (e) { diff --git a/dsh-mneme/test/content-hash.test.js b/dsh-mneme/test/content-hash.test.js new file mode 100644 index 00000000..cfaf06e8 --- /dev/null +++ b/dsh-mneme/test/content-hash.test.js @@ -0,0 +1,204 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { DatabaseSync } from "node:sqlite"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createStore } from "../src/store.js"; +import { contentHashOf, normalizeForHash } from "../src/content-hash.js"; + +// #254 内容哈希(写入准入的确定性计量锚)。 +// 这批用例锁三件事:归一化口径(只差格式不算两次)、派生列随内容走(insert/update/ +// CAS 三处都得重算,漏一处标记就在说谎)、以及老库打开时的迁移顺序(索引必须在加列 +// 之后建,否则 exec(SCHEMA) 直接报 no such column,插件整树加载失败)。 + +test("归一化口径:大小写/空白/标点/全角差异落在同一个键上", () => { + assert.equal(normalizeForHash("Hello, World!\n"), "hello world"); + assert.equal(normalizeForHash("hello world"), "hello world", "NFKC 折全角"); + assert.equal( + contentHashOf({ title: "Task A", content: "Line1.\n\nLine2" }), + contentHashOf({ title: "task a", content: "line1 line2" }) + ); + // 分隔符防拼接歧义:('ab','c') 与 ('a','bc') 不能同键 + assert.notEqual(contentHashOf({ title: "ab", content: "c" }), contentHashOf({ title: "a", content: "bc" })); + assert.notEqual(contentHashOf({ title: "t", content: "同一件事" }), contentHashOf({ title: "t", content: "另一件事" })); +}); + +test("归一化口径:同义词与词序不折叠(哈希只认逐字节相同)", () => { + assert.notEqual(contentHashOf({ title: "t", content: "a b" }), contentHashOf({ title: "t", content: "b a" })); +}); + +test("已知等价类:去标点让 3.5 与 35 同键(口径选择的结果,不是判据出错)", () => { + // 这条把「已知会误报」的样子钉住,防止有人把文件头的「命中即精确重复」当承诺、 + // 进而以为这是个 bug 去改归一化——改口径要带存量重算(见 content-hash.js 文件头)。 + assert.equal(contentHashOf({ title: "t", content: "版本 3.5" }), contentHashOf({ title: "t", content: "版本 35" })); + assert.equal(contentHashOf({ title: "t", content: "a-b" }), contentHashOf({ title: "t", content: "ab" })); + // 但数字本身没被折叠:不同数值不落同一键 + assert.notEqual(contentHashOf({ title: "t", content: "3.5" }), contentHashOf({ title: "t", content: "3.6" })); +}); + +test("空标题空正文没有可比内容 → null(不互相匹配成「全是重复」)", () => { + assert.equal(contentHashOf({ title: "", content: "" }), null); + assert.equal(contentHashOf({ title: " ", content: "\n\t" }), null); + assert.equal(contentHashOf({}), null); + assert.ok(contentHashOf({ title: "只有标题" })); +}); + +test("save 落 content_hash,findContentHashMatches 按 type + scope 收窄", () => { + const store = createStore(":memory:"); + const a = store.save({ type: "project", title: "同一件事", content: "正文", agent_scope: "A" }); + assert.equal(a.content_hash, contentHashOf({ title: "同一件事", content: "正文" })); + + const sameScope = store.findContentHashMatches({ type: "project", hash: a.content_hash, agent_scope: "A" }); + assert.deepEqual(sameScope.map((m) => m.id), [a.id]); + + // 跨 scope 永不互判(与 saveWithDedupe 的三维去重键同口径) + const otherScope = store.findContentHashMatches({ type: "project", hash: a.content_hash, agent_scope: "B" }); + assert.equal(otherScope.length, 0); + const unannotated = store.findContentHashMatches({ type: "project", hash: a.content_hash }); + assert.equal(unannotated.length, 0, "已标注行不与未标注行匹配"); + + // 入参先按 store 自己的口径归一,免得调用方传 " A " 就漏配 + assert.equal(store.findContentHashMatches({ type: "project", hash: a.content_hash, agent_scope: " A " }).length, 1); + // type 不同不互判 + assert.equal(store.findContentHashMatches({ type: "preference", hash: a.content_hash, agent_scope: "A" }).length, 0); + store.close(); +}); + +test("归档行在候选集内(#275 分界:出口止体积、不止重复)", () => { + const store = createStore(":memory:"); + const saved = store.save({ type: "project", title: "被归档的事实", content: "同样的正文" }); + store.setArchived(saved.id, true); + const hits = store.findContentHashMatches({ type: "project", hash: saved.content_hash }); + assert.deepEqual(hits, [{ id: saved.id, archived: true, forgotten: false }]); + store.close(); +}); + +test("同键命中超过窗口宽度时,活跃行仍在窗口内(归档动作会顶 updated_at)", () => { + // 回归(自动评审 #311 store.js:1055):LIMIT 先于调用方的「活区优先」执行。若只按 + // updated_at 倒序,第 11 条起写下的归档行会把更早的活跃行挤出窗口,调用方只能看见 + // 归档命中,把「活跃重复」错记成「归档重复」——正是 dup 信号要分流的两类。 + const store = createStore(":memory:"); + const live = store.save({ type: "project", title: "窗口内", content: "同一段正文" }); + assert.ok(live.content_hash, "写入即落 content_hash"); + const sinks = []; + for (let i = 0; i < 10; i++) { + sinks.push(store.save({ type: "project", title: "窗口内", content: "同一段正文" })); + } + for (const row of sinks) store.setArchived(row.id, true); + const hits = store.findContentHashMatches({ type: "project", hash: live.content_hash }); + assert.equal(hits.length, 10, "窗口宽度不变"); + assert.deepEqual(hits[0], { id: live.id, archived: false, forgotten: false }); + store.close(); +}); + +test("update 与 compareAndUpdate 都重算 content_hash", () => { + const store = createStore(":memory:"); + const saved = store.save({ type: "project", title: "T", content: "旧正文" }); + assert.equal(store.findContentHashMatches({ type: "project", hash: contentHashOf({ title: "T", content: "旧正文" }) }).length, 1); + + const updated = store.update(saved.id, { content: "新正文" }); + assert.equal(updated.content_hash, contentHashOf({ title: "T", content: "新正文" })); + assert.equal( + store.findContentHashMatches({ type: "project", hash: contentHashOf({ title: "T", content: "旧正文" }) }).length, + 0, + "旧内容的锚必须失效" + ); + + const cas = store.compareAndUpdate(updated.id, updated.updated_at, { title: "T2" }); + assert.equal(cas.content_hash, contentHashOf({ title: "T2", content: "新正文" })); + assert.equal( + store.findContentHashMatches({ type: "project", hash: contentHashOf({ title: "T", content: "新正文" }) }).length, + 0, + "CAS 路径漏重算的话标记会指向不再存在的内容" + ); + store.close(); +}); + +test("demoteToSummary 与 restoreContent 也重算 content_hash", () => { + const store = createStore(":memory:"); + const saved = store.save({ type: "project", title: "T", content: "完整正文" }); + const demoted = store.demoteToSummary(saved.id, "摘要行"); + assert.equal(demoted.content_hash, contentHashOf({ title: "T", content: "摘要行" })); + assert.equal( + store.findContentHashMatches({ type: "project", hash: contentHashOf({ title: "T", content: "完整正文" }) }).length, + 0, + "降级后的行装的是摘要,锚不能还指着完整正文" + ); + + const restored = store.restoreContent(saved.id); + assert.equal(restored.content_hash, contentHashOf({ title: "T", content: "完整正文" })); + store.close(); +}); + +test("老库打开:加列 + 建索引 + 存量回填(索引顺序错会直接开不起来)", () => { + const dir = mkdtempSync(join(tmpdir(), "dsh-mneme-hash-migrate-")); + const dbPath = join(dir, "legacy.db"); + try { + const legacy = new DatabaseSync(dbPath); + legacy.exec(`CREATE TABLE memories ( + id TEXT PRIMARY KEY, type TEXT NOT NULL, title TEXT NOT NULL, content TEXT NOT NULL, + tags TEXT NOT NULL DEFAULT '[]', importance INTEGER NOT NULL DEFAULT 3, + forgotten INTEGER NOT NULL DEFAULT 0, archived INTEGER NOT NULL DEFAULT 0, + source TEXT, created_at TEXT NOT NULL, updated_at TEXT NOT NULL + );`); + legacy + .prepare( + "INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, created_at, updated_at) VALUES (?, ?, ?, ?, '[]', 3, 0, ?, NULL, ?, ?)" + ) + .run("legacy-1", "project", "老库事实", "同一件事的正文", 0, "2026-01-01T00:00:00.000Z", "2026-01-01T00:00:00.000Z"); + legacy + .prepare( + "INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, created_at, updated_at) VALUES (?, ?, ?, ?, '[]', 3, 0, ?, NULL, ?, ?)" + ) + .run("legacy-2", "project", "被归档的老事实", "归档区里同一件事的正文", 1, "2026-01-02T00:00:00.000Z", "2026-01-02T00:00:00.000Z"); + legacy.close(); + + const store = createStore(dbPath); + const index = store.db + .prepare("SELECT name FROM sqlite_master WHERE type='index' AND name='idx_memories_content_hash'") + .get(); + assert.ok(index, "老库也要建上索引(候选集靠它 seek,否则每次写入都是全表扫)"); + + // 索引不只是「存在」:回填的 NULL 扫描与准入的候选集查询都得真的走它。列序写反 + // (type 打头)时前者退化成全表扫——每次打开都要把全部正文读一遍。 + const planOf = (sql, ...params) => + store.db.prepare(`EXPLAIN QUERY PLAN ${sql}`).all(...params).map((r) => r.detail).join(" | "); + assert.match( + planOf("SELECT id, title, content FROM memories WHERE content_hash IS NULL"), + /idx_memories_content_hash/, + "存量回填走索引" + ); + assert.match( + planOf("SELECT id FROM memories WHERE type = ? AND content_hash = ?", "project", "x"), + /idx_memories_content_hash/, + "候选集查询走索引" + ); + + const backfilled = store.getById("legacy-1"); + assert.equal(backfilled.content_hash, contentHashOf({ title: "老库事实", content: "同一件事的正文" })); + assert.deepEqual( + store.findContentHashMatches({ type: "project", hash: backfilled.content_hash }).map((m) => m.id), + ["legacy-1"] + ); + const archivedBackfilled = store.getById("legacy-2"); + assert.deepEqual( + store.findContentHashMatches({ type: "project", hash: archivedBackfilled.content_hash }), + [{ id: "legacy-2", archived: true, forgotten: false }], + "存量归档行回填后也要在候选集里(不然这条信号在老库上量不到归档区)" + ); + + // 回填后仍能正常写入(幂等迁移没有把表锁死) + const fresh = store.save({ type: "project", title: "新行", content: "新正文" }); + assert.ok(fresh.content_hash); + const firstPassHash = archivedBackfilled.content_hash; + store.close(); + + // 再开一次:只算 NULL 行,已有锚原样保留 + const again = createStore(dbPath); + assert.equal(again.getById("legacy-2").content_hash, firstPassHash, "重复打开不得改写已有锚"); + again.close(); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); diff --git a/dsh-mneme/test/write-admission.test.js b/dsh-mneme/test/write-admission.test.js index d24ef36d..e2f81162 100644 --- a/dsh-mneme/test/write-admission.test.js +++ b/dsh-mneme/test/write-admission.test.js @@ -103,6 +103,93 @@ test("g2:并入已有行不算同话题重写", () => { assert.equal(rows.length, 1, "合并是去重机制在正常工作,既不进预算也不产生 g2"); }); +// #254 内容哈希(exact duplicate):判据是归一化后逐字节相同,不是相似度;候选集把 +// 归档行一起纳入——归档行不在 saveWithDedupe 的候选集里(store.list 默认排除归档), +// 同一个事实再写一次就是新行,这正是这条信号要量的穿透。 +test("dup:命中归档行的重复写入被记下来(归档行在候选集内)", () => { + const { store, service } = setup(); + const first = service.saveWithDedupe({ type: "project", title: "归档事实", content: "同一件事的正文" }); + store.setArchived(first.memory.id, true); + const second = service.saveWithDedupe({ + _sessionKey: "sess-dup", + type: "project", + title: "归档事实", + content: "同一件事的正文" + }); + assert.equal(second.action, "created", "归档行不在去重候选集里,所以确实新建了一行"); + + const rows = admissionRows(store, "sess-dup"); + assert.equal(rows.length, 1); + assert.deepEqual(rows[0].metadata.dup, { memory_id: first.memory.id, archived: true, forgotten: false }); +}); + +test("dup:只差格式(大小写/空白/标点)的重复也算命中", () => { + const { store, service } = setup(); + const first = service.saveWithDedupe({ type: "project", title: "Task A", content: "Line1.\n\nLine2" }); + // 标题只差大小写 → saveWithDedupe 的精确标题匹配落空 → 走新建路径 + const second = service.saveWithDedupe({ + _sessionKey: "sess-fmt", + type: "project", + title: "task a", + content: "line1 line2" + }); + assert.equal(second.action, "created"); + const rows = admissionRows(store, "sess-fmt"); + assert.deepEqual(rows[0].metadata.dup, { memory_id: first.memory.id, archived: false, forgotten: false }); +}); + +test("dup:同内容既有活跃行又有归档行时报活跃那条(活区优先,而不是谁最近被改过)", () => { + // 回归点(单盲审查 L2):候选集按 updated_at 倒序返回,直接取第一条会把「命中归档 + // 区」这个信号盖掉(归档动作刚刷过 updated_at)。口径:有活跃命中就报它。 + const { store, service } = setup(); + const active = store.save({ type: "project", title: "Task A", content: "同一件事" }); + const archived = store.save({ type: "project", title: "task a", content: "同一件事" }); + store.setArchived(archived.id, true); + const created = service.saveWithDedupe({ + _sessionKey: "sess-both", type: "project", title: "TASK a", content: "同一件事" + }); + assert.equal(created.action, "created", "标题只差大小写 → 精确标题匹配落空,走新建路径"); + assert.deepEqual(admissionRows(store, "sess-both")[0].metadata.dup, { memory_id: active.id, archived: false, forgotten: false }); +}); + +test("dup:只剩已遗忘未归档的命中时,不把它记成活区命中", () => { + // 回归(自动评审 #311 write-admission.js:158):只回 archived 会把遗忘区命中写成 + // archived:false,读审计的人会以为存在活跃重复;而 saveWithDedupe 的候选集本来就排除 + // 遗忘行(store.list 默认 includeForgotten=false),这类命中同样是穿透。两个出口分字段记。 + // 两条行同标题同正文:遗忘之后标题候选集里已经没有它,第二次写入自然走新建路径。 + const { store, service } = setup(); + const first = service.saveWithDedupe({ type: "project", title: "遗忘事实", content: "同一段遗忘正文" }); + store.setForget(first.memory.id, true); + const second = service.saveWithDedupe({ + _sessionKey: "sess-forgot", type: "project", title: "遗忘事实", content: "同一段遗忘正文" + }); + assert.equal(second.action, "created", "遗忘行不在去重候选集里,所以确实新建了一行"); + assert.deepEqual(admissionRows(store, "sess-forgot")[0].metadata.dup, { + memory_id: first.memory.id, archived: false, forgotten: true + }); +}); + +test("dup:不同内容不记 dup(负样本,免得信号恒真)", () => { + const { store, service } = setup(); + service.saveWithDedupe({ _sessionKey: "s", type: "project", title: "T1", content: "第一件事的正文" }); + service.saveWithDedupe({ _sessionKey: "s", type: "project", title: "T2", content: "另一件事的正文" }); + const rows = admissionRows(store, "s"); + assert.equal(rows.length, 2); + for (const row of rows) assert.equal(row.metadata.dup, undefined); +}); + +test("dup:pinned 行不查候选集(整个不在闸门里,与 g2 的不当基准同口径)", () => { + const { store, service } = setup(); + service.saveWithDedupe({ _sessionKey: "s", type: "constraint", title: "边界 X", content: "同一段边界" }); + service.saveWithDedupe({ _sessionKey: "s", type: "constraint", title: "边界 x", content: "同一段边界" }); + const rows = admissionRows(store, "s"); + assert.equal(rows.length, 2); + for (const row of rows) { + assert.equal(row.metadata.exempt, "pinned"); + assert.equal(row.metadata.dup, undefined, "穿透口不发候选集查询,也不记 dup"); + } +}); + test("话题表由审计行重建:新实例(进程重启)仍能认出同话题", () => { const { store, service } = setup(); service.saveWithDedupe({ _sessionKey: "s", type: "project", title: "A", content: "#99 的记录" });