Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 54 additions & 0 deletions dsh-mneme/lib/content-hash.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
// #254 写入准入的确定性计量锚:内容归一化哈希(exact duplicate 判定)。
//
// 为什么用哈希而不是相似度:计量阶段要的是一个零成本、可复算、不留争议的判据。
// 相似度阈值两边都是误判——同一件事换个说法就掉到线下,不同的事共享话题词又顶到
// 线上,阈值往哪边挪都只是换一种错法(#254 拍板:计量信号用内容哈希)。哈希只认
// 「归一化后逐字节相同」,算不出来就只有一种解释。
//
// 但它不等于「原文逐字节相同」:归一化会折掉格式,也会折掉标点,于是 `版本 3.5` 与
// `版本 35`、`a-b` 与 `ab` 落同一个键。折叠范围(大小写 / 空白 / 标点)是口径本身
// 定的,#254 已拍板——误报面在计量阶段可接受(只标记、不拦截),但「命中即精确重复」
// 这句话别升级成「命中即内容相同」。要更窄的判据就得改归一化口径,而那要带存量重算。
//
// 归一化口径一次定死(NFKC → 小写 → 去标点 → 空白折叠 → trim),动机是让「只差
// 格式」的两次写入落在同一个键上:全角/半角、大小写、行内换行与缩进、句末标点都
// 不该让同一件事变成两条。反过来它不折叠同义词、不排序词序——那些只能靠相似度,
// 而相似度已判定不可用于本信号。
//
// 改口径必须带存量重算策略:memories.content_hash 是派生列,口径一变,库里已有的
// 值全部按旧口径算——补 NULL 修不好,只能整列重算(UPDATE 全表 + 重建索引)。
// 所以口径只能在这里改,且改要单独成批,不要混进别的改动里。
//
// 空内容(标题与正文归一后都为空)返回 null:这类写入没有可比内容,落 NULL 让它们
// 不互相匹配成「全是重复」。
//
// 另有一处形状相近、口径不同的哈希:service.js 里镜像人工编辑的 digest 校验(两处)
// 用 sha256(title\0content),那里**不**归一化(要逐字节保真)。两把哈希用途不同,别互换。
import { createHash } from "node:crypto";

/**
* 归一化一段文本用于内容比对(口径见文件头)。
* @param {unknown} value
* @returns {string}
*/
export function normalizeForHash(value) {
return String(value ?? "")
.normalize("NFKC")
.toLowerCase()
.replace(/\p{P}/gu, "")
.replace(/\s+/g, " ")
.trim();
}

/**
* 一行记忆的内容锚:标题与正文分别归一后用 NUL 拼接再取 sha256(十六进制全串)。
* NUL 分隔是防拼接歧义——('ab','c') 与 ('a','bc') 不能落到同一个键上。
* @param {{title?: string, content?: string}|null|undefined} memory
* @returns {string|null} 归一后无内容时返回 null
*/
export function contentHashOf(memory) {
const title = normalizeForHash(memory?.title);
const content = normalizeForHash(memory?.content);
if (!title && !content) return null;
return createHash("sha256").update(`${title}\u0000${content}`).digest("hex");
}
110 changes: 96 additions & 14 deletions dsh-mneme/lib/store.js
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import { DatabaseSync } from "node:sqlite";
import { randomUUID } from "node:crypto";
import { statSync } from "node:fs";
import { contentHashOf } from "./content-hash.js";

const SCHEMA = `
CREATE TABLE IF NOT EXISTS memories (
Expand All @@ -14,6 +15,7 @@ CREATE TABLE IF NOT EXISTS memories (
archived INTEGER NOT NULL DEFAULT 0,
source TEXT,
content_history TEXT,
content_hash TEXT,
embedding TEXT,
epistemic_status TEXT NOT NULL DEFAULT 'subjective',
last_accessed_at TEXT,
Expand Down Expand Up @@ -407,6 +409,7 @@ function toRow(row) {
type: row.type,
title: row.title,
content: row.content,
content_hash: row.content_hash ?? undefined,
tags: parseTags(row.tags),
importance: row.importance,
forgotten: row.forgotten === 1,
Expand Down Expand Up @@ -715,6 +718,34 @@ export function createStore(path) {
addColumn("memories", "workspace_scope_source", "ALTER TABLE memories ADD COLUMN workspace_scope_source TEXT");
addColumn("memories", "scope_decided_at", "ALTER TABLE memories ADD COLUMN scope_decided_at TEXT");

// #254 计量信号:内容归一化哈希(口径与动机写在 content-hash.js 的文件头)。派生
// 列,不参与任何判定——只给「同一内容又被写了一次」当等值锚。索引建在加列之后:
// 老库打开时 SCHEMA 的 CREATE TABLE 对既有表不生效,列还不存在,把索引写进 SCHEMA
// 会直接报 no such column(与下方 llm_audit_logs.session_key 同理)。
addColumn("memories", "content_hash", "ALTER TABLE memories ADD COLUMN content_hash TEXT");
// 列序是 content_hash 打头:哈希几乎唯一,等值 seek 就已收窄到候选行,type 只当同
// 一次 seek 里的第二列过滤;反过来(type 打头)下面的存量回填就得全表扫——每次打开
// 都要把全部正文读一遍。
db.exec("CREATE INDEX IF NOT EXISTS idx_memories_content_hash ON memories(content_hash, type)");
// 存量回填:老库(含归档区)的行都没有值,而「归档行纳入去重候选集」这条口径正是
// 要靠既有归档行出数据。只算 NULL 行、幂等;改口径不是补 NULL 能修好的,要整列重算
// (见 content-hash.js 的文件头)。空标题空正文的行算不出锚,跳过不写——否则每次
// 打开都要为这些行再跑一遍 UPDATE。
{
const updates = db
.prepare("SELECT id, title, content FROM memories WHERE content_hash IS NULL")
.all()
.map((row) => [contentHashOf(row), row.id])
.filter(([hash]) => hash);
if (updates.length) {
const stmt = db.prepare("UPDATE memories SET content_hash = ? WHERE id = ? AND content_hash IS NULL");
// 一个事务包住整批:逐条自动提交是每行一次 WAL 提交,几千行的存量库上纯属浪费。
runAtomically(() => {
for (const [hash, id] of updates) stmt.run(hash, id);
});
}
}

// Legacy dream_runs without policy_epoch → backfill with the default epoch.
addColumn("dream_runs", "policy_epoch", "ALTER TABLE dream_runs ADD COLUMN policy_epoch INTEGER NOT NULL DEFAULT 0");
addColumn("dream_runs", "run_type", "ALTER TABLE dream_runs ADD COLUMN run_type TEXT NOT NULL DEFAULT 'auto'");
Expand Down Expand Up @@ -953,6 +984,8 @@ export function createStore(path) {
const importance = Number.isInteger(memory.importance) ? memory.importance : 3;
const evidence = Array.isArray(memory.evidence) ? JSON.stringify(memory.evidence) : null;
const docPath = typeof memory.doc_path === "string" && memory.doc_path.trim() ? memory.doc_path : null;
// #254 计量锚:由本行的 title/content 派生,调用方传什么都以这里算出的为准。
const contentHash = contentHashOf(memory);
const embedding = Array.isArray(memory.embedding) && memory.embedding.length
? JSON.stringify(memory.embedding)
: null;
Expand All @@ -963,8 +996,8 @@ export function createStore(path) {
: inferEpistemicStatus(memory);
runAtomically(() => {
db.prepare(
`INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, content_history, quality_score, embedding, epistemic_status, agent_scope, workspace_scope, agent_scope_source, workspace_scope_source, scope_decided_at, sensitivity, occurred_at, evidence, doc_path, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`
`INSERT INTO memories (id, type, title, content, tags, importance, forgotten, archived, source, content_history, quality_score, embedding, epistemic_status, agent_scope, workspace_scope, agent_scope_source, workspace_scope_source, scope_decided_at, sensitivity, occurred_at, evidence, doc_path, content_hash, created_at, updated_at)
VALUES (?, ?, ?, ?, ?, ?, 0, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`
).run(
id,
type,
Expand All @@ -987,6 +1020,7 @@ export function createStore(path) {
normalizeOccurredAt(memory.occurred_at),
evidence,
docPath,
contentHash,
now,
now
);
Expand All @@ -999,6 +1033,41 @@ export function createStore(path) {
return getById(id);
}

/**
* #254 计量:与本次写入「归一化内容哈希」相同的既有行(去重候选集)。
* 结构化锚先收窄成本:按哈希等值 seek(走 idx_memories_content_hash,哈希几乎唯
* 一,落在候选集上的行只有几条),再按 type 与 scope 三维过滤——不是全表扫、也不
* 逐行比哈希。scope 用 IS 比(NULL 安全):未标注行互相匹配、与已标注行不匹配,
* 与 service.js 去重键(scopeKeyOf)同口径;入参先按 store 自己的 normalizeScopeText
* 归一,免得调用方传 " foo " 就漏配。
* 归档行与已遗忘行都在集内:#275 的分界是「出口止体积、不止重复」——被质量闸归档
* 的同一个事实必须仍能判成重复,否则同样的内容再写一次又是一条新行(#254 拍板:
* 归档行纳入去重候选集)。
* 排序即取舍:活跃行排在归档/遗忘行之前,再按 updated_at 倒序。LIMIT 先于调用方的
* 「活区优先」判断执行,而归档动作本身会顶 updated_at——纯按时间倒序时,同键命中一
* 旦超过窗口宽度,活跃行就被归档行挤出候选集,调用方只能看到归档命中,把「活跃重复」
* 误报成「归档重复」,恰好把这条信号要分流的两类弄反。
* @returns {Array<{id: string, archived: boolean, forgotten: boolean}>} 活跃行优先,其后按最近写入
*/
function findContentHashMatches({ type, hash, agent_scope: agentScope, workspace_scope: workspaceScope, sensitivity, limit = 10 } = {}) {
if (!hash || !type) return [];
const lim = Number.isInteger(limit) && limit > 0 ? Math.min(limit, 50) : 10;
const rows = db.prepare(
`SELECT id, archived, forgotten FROM memories
WHERE type = ? AND content_hash = ?
AND agent_scope IS ? AND workspace_scope IS ? AND sensitivity IS ?
ORDER BY archived ASC, forgotten ASC, updated_at DESC, id LIMIT ?`
).all(
type,
hash,
normalizeScopeText(agentScope),
normalizeScopeText(workspaceScope),
normalizeScopeText(sensitivity),
lim
);
return rows.map((row) => ({ id: row.id, archived: row.archived === 1, forgotten: row.forgotten === 1 }));
}

function update(id, patch) {
const existing = getById(id);
if (!existing) throw new Error(`memory not found: ${id}`);
Expand Down Expand Up @@ -1045,13 +1114,18 @@ export function createStore(path) {
const nextDecidedAt = patch.scope_decided_at !== undefined
? normalizeOccurredAt(patch.scope_decided_at)
: (existing.scope_decided_at ?? null);
// #254:标题或正文变了,锚跟着重算——它是当前内容的派生物,留旧值等于让「这行
// 现在装的什么」和标记对不上。
const nextTitle = patch.title ?? existing.title;
const nextContent = patch.content ?? existing.content;
const contentHash = contentHashOf({ title: nextTitle, content: nextContent });
runAtomically(() => {
db.prepare(
`UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, agent_scope=?, workspace_scope=?, agent_scope_source=?, workspace_scope_source=?, scope_decided_at=?, evidence=?, updated_at=? WHERE id=?`
`UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, agent_scope=?, workspace_scope=?, agent_scope_source=?, workspace_scope_source=?, scope_decided_at=?, evidence=?, content_hash=?, updated_at=? WHERE id=?`
).run(
type,
patch.title ?? existing.title,
patch.content ?? existing.content,
nextTitle,
nextContent,
JSON.stringify(patch.tags ?? existing.tags),
Number.isInteger(patch.importance) ? patch.importance : existing.importance,
patch.source !== undefined ? patch.source : (existing.source ?? null),
Expand All @@ -1065,6 +1139,7 @@ export function createStore(path) {
nextWorkspaceSource,
nextDecidedAt,
evidence,
contentHash,
now,
id
);
Expand Down Expand Up @@ -1126,22 +1201,27 @@ export function createStore(path) {
// dirty == false — recoverMirror sees no debt and the mirror stays stale.
// Wrapping both in one transaction means a CAS miss rolls back cleanly too
// (no write, no generation bump).
// 同 update:#254 的锚随标题/正文重算(CAS 路径也改这两列)。
const nextTitle = patch.title ?? existing.title;
const nextContent = patch.content ?? existing.content;
const contentHash = contentHashOf({ title: nextTitle, content: nextContent });
let applied = false;
runAtomically(() => {
const result = db.prepare(
`UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, updated_at=?
`UPDATE memories SET type=?, title=?, content=?, tags=?, importance=?, source=?, content_history=?, quality_score=?, embedding=?, epistemic_status=?, content_hash=?, updated_at=?
WHERE id=? AND updated_at=?`
).run(
type,
patch.title ?? existing.title,
patch.content ?? existing.content,
nextTitle,
nextContent,
JSON.stringify(patch.tags ?? existing.tags),
Number.isInteger(patch.importance) ? patch.importance : existing.importance,
patch.source !== undefined ? patch.source : (existing.source ?? null),
contentHistory,
qualityScore,
embedding,
epistemicStatus,
contentHash,
now,
id,
expectedUpdatedAt
Expand Down Expand Up @@ -1193,15 +1273,15 @@ export function createStore(path) {
function demoteToSummary(id, summary, { minRefTimeMs } = {}) {
let changed = false;
runAtomically(() => {
const row = db.prepare("SELECT last_accessed_at, content, _full_content FROM memories WHERE id = ?").get(id);
const row = db.prepare("SELECT title, last_accessed_at, content, _full_content FROM memories WHERE id = ?").get(id);
if (!row || row._full_content) return;
if (minRefTimeMs !== undefined && row.last_accessed_at) {
const lastMs = Date.parse(row.last_accessed_at);
if (lastMs >= minRefTimeMs) return; // touched after snapshot — still hot
}
db.prepare(
"UPDATE memories SET content = ?, _full_content = ?, updated_at = ? WHERE id = ?"
).run(summary, row.content, nowIso(), id);
"UPDATE memories SET content = ?, _full_content = ?, content_hash = ?, updated_at = ? WHERE id = ?"
).run(summary, row.content, contentHashOf({ title: row.title, content: summary }), nowIso(), id);
incrementGeneration();
changed = true;
});
Expand All @@ -1212,11 +1292,11 @@ export function createStore(path) {
function restoreContent(id) {
let changed = false;
runAtomically(() => {
const row = db.prepare("SELECT content, _full_content FROM memories WHERE id = ?").get(id);
const row = db.prepare("SELECT title, content, _full_content FROM memories WHERE id = ?").get(id);
if (!row || !row._full_content) return;
db.prepare(
"UPDATE memories SET content = ?, _full_content = NULL, updated_at = ? WHERE id = ?"
).run(row._full_content, nowIso(), id);
"UPDATE memories SET content = ?, _full_content = NULL, content_hash = ?, updated_at = ? WHERE id = ?"
).run(row._full_content, contentHashOf({ title: row.title, content: row._full_content }), nowIso(), id);
incrementGeneration();
changed = true;
});
Expand Down Expand Up @@ -2606,6 +2686,8 @@ export function createStore(path) {
saveDocument,
update,
compareAndUpdate,
// #254:写入准入的内容哈希候选集(只读)。
findContentHashMatches,
remove,
setForget,
setArchived,
Expand Down
Loading
Loading