diff --git a/dsh-mneme/docs/STORAGE.md b/dsh-mneme/docs/STORAGE.md index ee1e00a1..b3afb918 100644 --- a/dsh-mneme/docs/STORAGE.md +++ b/dsh-mneme/docs/STORAGE.md @@ -65,6 +65,25 @@ VACUUM 是 O(库大小),且需要排他写锁:库越大越慢,期间的写 - 窗口内(默认 7 天)的输入快照不动:近期 run 仍可能在离线回放里被用到。 - `run_type='organize'` 的行任何窗口都不清:它们的 `input` 存的是 apply 的重放载荷(`src/organize.js` 直接读它重建候选索引),不是可重建的快照。置空会让那次 apply 静默空转并把报告锁死。 +## 升格吸收的 evidence(#230 × #275) + +`memory_register_document` 注册成功后,这份文档引用的 evidence 行会**在同一事务里**翻成归档(absorbed)。一次吸收 20 条,活跃面就少 20 行——而不是「21 行不是 1 行」。 + +- 只翻标志位:内容、`content_history`、审计行一条不删,随时可以还原; +- `constraint` / `preference` 永不自动归档(#249 的逐字保真池),要归档得手动来; +- 只吸收**原子条**:别的 `document` 行与 `summary` 行不碰——前者本体是磁盘上的文件(归档它会让同一路径下次注册留下第二行),后者是 dream 总览与叙述(按 source 身份去重、按注入档位常驻); +- 想保留活跃面:工具带 `keep_evidence_active: true`(注册器参数是 `archiveEvidence: false`); +- 重注册同一份文档(出新版)时,那些已归档的 evidence 行仍可为**它所属的那份文档**背书;换个文档拿归档 id 当证据,照旧按捏造拒绝。 + +## 第五指标:归档净增速率 + 可压掉行数 + +面板「记忆复用」卡与 `GET /api/dsh-mneme/recall-stats` 的返回里多了一个 `archive` 块(#275 拍板:与既有指标同位): + +- `total` / `addedInWindow` / `perDay`:归档区现有行数、窗口内新进的、以及日均净增速率。归档时刻取行的 `updated_at`(`setArchived` 会刷它)——代理口径:再改一次已归档行就会被算进本窗口,精确口径要一个 `archived_at` 列(属第二批题材); +- `compressible.rows` / `.groups`:**可压掉行数**——同 type、同 scope、内容归一化后逐字节相同的归档行里,每组留下的那一行之外的那些。判据用内容哈希(与写入准入同一把尺),**不**建在向量近重复上:回收动作本身就会清掉归档行的向量,指标不能指望自己的输入还在。 + +这是观察口,不是动作触发:真删仍属第二批(未立项)。 + ## 与 #254 的分界 本入口止体积、不止重复。归档区里同主题条目堆积是**出口**问题,归 #254 的写入准入与后续整理动作面;这里一条记忆都不碰(维护者拍板 4)。 diff --git a/dsh-mneme/lib/client.js b/dsh-mneme/lib/client.js index 57a46578..18fc352d 100644 --- a/dsh-mneme/lib/client.js +++ b/dsh-mneme/lib/client.js @@ -559,6 +559,7 @@ window.__ModuleLoader__.load({ "memory.status.recallStats": "记忆复用", "memory.status.recallHint": "复用率 {rate} · 僵尸 {zombie}/{active}(豁免 {exempt})· 30 天回执 {runs} 次 · Top:{top}", "memory.status.recallInject": "注入 {runs} 轮 {count} 条,槽位填充 {fill}%", + "memory.status.recallArchive": "归档 {total} 行({add}/天),精确重复可压掉 {compress} 行", "memory.status.viewAll": "查看全部", "memory.status.depositedCount": "沉淀的记忆({n})", "memory.status.archivedCount": "已归档的记忆({n})", @@ -946,6 +947,7 @@ window.__ModuleLoader__.load({ "memory.status.recallStats": "Recall reuse", "memory.status.recallHint": "reuse {rate} · zombie {zombie}/{active} (exempt {exempt}) · {runs} runs in 30d · top: {top}", "memory.status.recallInject": "injected {runs} turns · {count} items ({fill}% slots)", + "memory.status.recallArchive": "archived {total} rows (+{add}/day) · {compress} exactly-duplicate rows compressible", "memory.status.viewAll": "View all", "memory.status.depositedCount": "Deposited memories ({n})", "memory.status.archivedCount": "Archived memories ({n})", @@ -3613,7 +3615,7 @@ window.__ModuleLoader__.load({ // 口径不可信)或库为空时整卡不渲染,前端不感知;truncated(扫描超上限) // 只影响 hint 里的回执计数,不挡渲染。 function RecallStatsCard({ t }) { - const [state, setState] = useState({ loading: true, off: false, rate: null, zombie: 0, active: 0, exempt: 0, runs: 0, top: "", inject: "" }); + const [state, setState] = useState({ loading: true, off: false, rate: null, zombie: 0, active: 0, exempt: 0, runs: 0, top: "", inject: "", archive: "" }); useEffect(() => { let cancelled = false; apiFetch("/api/dsh-mneme/recall-stats?window=30") @@ -3621,7 +3623,11 @@ window.__ModuleLoader__.load({ .then((d) => { if (cancelled) return; const z = d.zombie || {}; - const hasData = (d.coverage?.runsScanned ?? 0) > 0 || (z.activeCount ?? 0) > 0; + // 归档侧第五指标(#275)不能只靠 run/active 两路撑整张卡:整库归档是它的正常 + // 工作状态,那种库「窗口内没有召回回执、也没有活跃僵尸行」时若 hasData 为假, + // 组件直接走 off 分支 return null,归档指标永远没机会显示。 + const hasData = (d.coverage?.runsScanned ?? 0) > 0 || (z.activeCount ?? 0) > 0 + || (d.archive?.total ?? 0) > 0; if (!hasData) { setState({ loading: false, off: true }); return; @@ -3638,10 +3644,19 @@ window.__ModuleLoader__.load({ .replace("{count}", String(inj.injectedCount ?? 0)) .replace("{fill}", inj.slotFillRate == null ? "—" : String(Math.round(inj.slotFillRate * 100))) : ""; + // 第五指标(#275):归档净增速率 + 可压掉行数。与注入口径同款自门控 + // ——归档区为空时整段省略,不往卡里塞一个恒 0 的数字。 + const ar = d.archive || {}; + const archive = (ar.total ?? 0) > 0 + ? t("memory.status.recallArchive") + .replace("{total}", String(ar.total ?? 0)) + .replace("{add}", String(ar.perDay ?? 0)) + .replace("{compress}", String(ar.compressible?.rows ?? 0)) + : ""; setState({ loading: false, off: false, rate, zombie: z.zombieCount ?? 0, active: z.activeCount ?? 0, - exempt: z.exemptCount ?? 0, runs: d.coverage?.runsScanned ?? 0, top, inject + exempt: z.exemptCount ?? 0, runs: d.coverage?.runsScanned ?? 0, top, inject, archive }); }) .catch(() => { if (!cancelled) setState({ loading: false, off: true }); }); @@ -3660,7 +3675,7 @@ window.__ModuleLoader__.load({ .replace("{active}", String(state.active)) .replace("{exempt}", String(state.exempt)) .replace("{runs}", String(state.runs)) - .replace("{top}", state.top || "—") + (state.inject ? " · " + state.inject : "") + .replace("{top}", state.top || "—") + (state.inject ? " · " + state.inject : "") + (state.archive ? " · " + state.archive : "") }); } diff --git a/dsh-mneme/lib/document.js b/dsh-mneme/lib/document.js index f89d572a..79bf2d8c 100644 --- a/dsh-mneme/lib/document.js +++ b/dsh-mneme/lib/document.js @@ -60,20 +60,34 @@ export function isManagedDocumentPath(dir, path) { // id + 旧行 content_history 里的旧摘要(source=superseded)。 const supersededByPointer = (id) => `\n\n[superseded by ${id}]`; +// 升格吸收不碰的类型(#275 拍板 5 的边界):吸收的对象是「原子条」,而下面这两种 +// 行各有自己的生命周期,被归档会各自留下第二行—— +// document:本体是磁盘上的文件。归档它等于替用户撤下外档;而且 supersede 探测 +// 只看活跃行(store.list 默认排除归档),同一 path 下次注册会在旧行还挂着 +// 「新版本」语义时再铸一行,指针注记与 content_history 记账也走不到。 +// summary:dream 总览与叙述按 source 身份去重、并按注入档位常驻。归档它会让下一 +// 次做梦把它当不存在(候选集同样只取活跃行),于是库里多出一行同源总览。 +// 口径是保守方向的:拿不准的行一律不吸收——少做一步没有代价,收错一步是内容事故。 +const ABSORB_EXEMPT_TYPES = new Set(["document", "summary"]); + /** * Build the document registrar. Injected deps are service.js closure members * (store / embedQuery / transaction / finalize) so this module stays free of * service-internal wiring; `finalize` is the single write epilogue (mirror * sync + notify + re-embed) so callers never duplicate it. * - * @returns {(payload: object) => Promise} + * `pinnedTypes` 是注入的 #249 逐字保真池(service.js 的 PINNED_MEMORY_TYPES): + * 这里不 import 它在 service.js 里,是因为 service.js 反向 import 本模块—— + * 方向反了会成环。 + * + * @returns {(payload: object, opts?: object) => Promise} * `{ action: "created"|"superseded", memory, superseded?, evidence_kept, - * evidence_dropped, degraded }` + * evidence_dropped, evidence_archived, degraded }` * @throws flag 关 / 路径非法或文件缺失 / title 或 summary 为空 / evidence 全部 * 捏造 / 仅 vector 近重复(疑似重复外档,交 agent 裁决而不是静默二选一)。 */ -export function createDocumentRegistrar({ store, config, embedQuery, pushContentHistory, transaction, finalize }) { - return async function registerDocument(payload, { hiddenEvidenceIds = [] } = {}) { +export function createDocumentRegistrar({ store, config, embedQuery, pushContentHistory, transaction, finalize, pinnedTypes }) { + return async function registerDocument(payload, { hiddenEvidenceIds = [], archiveEvidence = true } = {}) { // opt-in 总闸(#230 对齐 #228 形态):关 = 注册入口整体不存在,错误信息 // 指回配置键,agent 能准确转告用户去开。 if (config?.documentMemoryEnabled !== true) { @@ -121,15 +135,14 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent const hidden = new Set((Array.isArray(hiddenEvidenceIds) ? hiddenEvidenceIds : []).map((id) => String(id))); const kept = []; const dropped = []; + // 存在但已归档的引用另记一份:它可能是被上一版文档升格吸收走的行(见下方 + // supersede 目标探测后的「认回」),不能一律按「不可用」处理。 + const archivedRefs = new Set(); for (const id of wanted) { const row = hidden.has(id) ? null : store.getById(id); + if (row?.archived) archivedRefs.add(id); (row && !row.archived ? kept : dropped).push(id); } - if (wanted.length > 0 && kept.length === 0) { - throw new Error( - `registerDocument: all ${wanted.length} evidence ids are unknown, archived or out of scope — fabricated evidence is rejected` - ); - } const title = String(payload?.title ?? "").trim(); if (!title) throw new Error("registerDocument: title is required"); @@ -162,6 +175,25 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent .filter((m) => scopeMatches(m)); const explicitTarget = scopeMatched.find((m) => samePath(m.doc_path, expanded)) ?? scopeMatched.find((m) => m.title.trim() === title); + // 升格吸收过的引用认回(#275 拍板 5 的配套):evidence 行在文档注册成功后就翻了 + // archived,于是「同一份文档出新版、evidence 照旧」这一最常规的路径会整批落在 + // dropped——那会被下面的捏造判据误报成「证据都是编的」。口径:已归档行仍可为 + // **吸收它的那份文档**背书(loser.evidence 就是它吸收走的名单),不为别的文档 + // 背书;捏造与跨 scope 照旧拒绝。 + if (explicitTarget) { + const own = new Set(explicitTarget.evidence ?? []); + for (const id of archivedRefs) { + if (!own.has(id)) continue; + const at = dropped.indexOf(id); + if (at >= 0) dropped.splice(at, 1); + kept.push(id); + } + } + if (wanted.length > 0 && kept.length === 0) { + throw new Error( + `registerDocument: all ${wanted.length} evidence ids are unknown, archived or out of scope — fabricated evidence is rejected` + ); + } if (!explicitTarget) { // vector 档:embedder 不可用/向量缺失一律跳过——去重是增强不是写入依赖 // (findSessionDuplicate 同原则)。probe 用 title+summary,与行向量 @@ -237,7 +269,25 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent store.setArchived(loser.id, true); superseded = store.getById(loser.id); } - return { created, superseded }; + // #275 拍板 5(#230 验收口径的补丁):升格吸收的 evidence 行随之退出活跃面—— + // 一次吸收 20 条,库里就是「21 行不是 1 行」。四条口径:①与本行同一事务(要么 + // 都成、要么都不成);②pinned(constraint / preference,即 #249 的逐字保真池) + // 永不自动归档——升格不能绕过它;③自有生命周期的类型(document / summary,见 + // ABSORB_EXEMPT_TYPES)不吸收,吸收的是原子条;④只翻标志位、内容与审计全留 + // (可恢复),opt-out 走 archiveEvidence。 + // pinnedTypes 拿不到就不做这一步:宁可少做,也不能把保真池当普通行收走。 + let evidenceArchived = 0; + if (archiveEvidence && pinnedTypes && typeof pinnedTypes.has === "function") { + for (const id of kept) { + // 事务内重读:期间被别的进程归档/遗忘的行不重复计数,也不误伤 pinned。 + const row = store.getById(id); + if (!row || row.archived || row.forgotten) continue; + if (pinnedTypes.has(row.type) || ABSORB_EXEMPT_TYPES.has(row.type)) continue; + store.setArchived(id, true); + evidenceArchived += 1; + } + } + return { created, superseded, evidenceArchived }; }); finalize(result.superseded ? [result.created, result.superseded] : [result.created]); return { @@ -246,6 +296,7 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent ...(result.superseded ? { superseded: result.superseded } : {}), evidence_kept: kept.length, evidence_dropped: dropped.length, + evidence_archived: result.evidenceArchived, degraded: dropped.length > 0 }; }; diff --git a/dsh-mneme/lib/recall-stats.js b/dsh-mneme/lib/recall-stats.js index 70faaa86..8721b6e0 100644 --- a/dsh-mneme/lib/recall-stats.js +++ b/dsh-mneme/lib/recall-stats.js @@ -5,6 +5,8 @@ // 新内聚块从第一行就落在自己的文件里,service.js 保留 barrel 出口、调用方零改动。 // 纯读不写,绝不触发任何 write hook。 +import { scopeKeyOf } from "./scope.js"; + /** * 口径(issue #217 评论 2026-09-18,锚 5bd2dab): * - Top-N:窗口内 recall_runs.candidates(最终返回集)按 id 计数,join @@ -20,6 +22,17 @@ * - coverage:recallRecordDefault 开启前的窗口算不到,earliestRunAt 为 * null(窗口内无回执)或早于窗口起点时,前端标注可信度;扫描行数有 * 上限,超出标 truncated(degraded 口径),不静默少算。 + * - archive(#275 拍板的第五指标,与上面几项同位): + * · total = 归档区现有行数;addedInWindow / perDay = 窗口内进入归档区的行数 + * 与其日均(净增速率)。归档时刻只能取 updated_at(setArchived 翻标志位时刷 + * 它)——这是代理口径:对已归档行做一次 memory_update,或对同一行重复调 + * setArchived(…, true),都会被算成「本窗口新进归档」,perDay 因此偏高。精确 + * 口径要一个 archived_at 列(schema 变更,属第二批题材),本版不加;真正的 + * 「净」增本来也要跨快照比 total(物理删除落地后,差值才会由负向变化体现)。 + * · compressible = 可压掉行数:同 type、同 scope 三维且内容哈希完全相同的归档 + * 行里,多出来的那些(每组留一行)。用内容哈希而不是向量近重复,是因为回收 + * 动作本身会清掉归档行向量(clearArchivedEmbeddings)——指标不能建在它自己 + * 的输入会被回收掉的数据上;精确重复与 #254 的计量口径同一把尺。 * * @param {object} store - createStore 产物(只调用 listRecallRunsSince / all) * @param {{ windowDays?: number, exemptDays?: number }} [options] @@ -36,7 +49,8 @@ export function recallStats(store, options = {}) { const rawSlots = Number(options.maxInjectSlots ?? 5); const maxInjectSlots = Number.isInteger(rawSlots) && rawSlots >= 1 ? rawSlots : 5; const now = Date.now(); - const since = new Date(now - windowDays * 86400000).toISOString(); + const sinceMs = now - windowDays * 86400000; + const since = new Date(sinceMs).toISOString(); const { rows: runs, total } = store.listRecallRunsSince(since); const hits = new Map(); // id -> { count, title, source } @@ -67,8 +81,29 @@ export function recallStats(store, options = {}) { let activeCount = 0; let zombieCount = 0; let exemptCount = 0; + // 第五指标(#275):归档区一侧单独累计,分组键与 #254 的去重候选集同构 + // (type + 哈希 + scope 三维)——可压掉的必须是「本来就会被判成同一件事」的行。 + let archivedTotal = 0; + let archivedInWindow = 0; + const hashGroups = new Map(); for (const m of memories) { - if (m.archived || m.forgotten) continue; + if (m.archived) { + archivedTotal += 1; + const archivedMs = Date.parse(m.updated_at ?? ""); + if (Number.isFinite(archivedMs) && archivedMs >= sinceMs) archivedInWindow += 1; + if (m.content_hash) { + const key = [ + m.type ?? "", + m.content_hash, + scopeKeyOf(m.agent_scope) ?? "", + scopeKeyOf(m.workspace_scope) ?? "", + scopeKeyOf(m.sensitivity) ?? "" + ].join("\u0000"); + hashGroups.set(key, (hashGroups.get(key) ?? 0) + 1); + } + continue; + } + if (m.forgotten) continue; const createdMs = Date.parse(m.created_at ?? ""); // created_at 解析不了时无法证明已过机会期 → 归入豁免,宁漏勿误伤 if (!Number.isFinite(createdMs) || now - createdMs < EXEMPT_MS) { @@ -98,6 +133,14 @@ export function recallStats(store, options = {}) { .sort((a, b) => b.count - a.count || a.id.localeCompare(b.id)) .slice(0, 10); + let compressibleGroups = 0; + let compressibleRows = 0; + for (const n of hashGroups.values()) { + if (n < 2) continue; + compressibleGroups += 1; + compressibleRows += n - 1; + } + return { windowDays, generatedAt: new Date(now).toISOString(), @@ -113,6 +156,13 @@ export function recallStats(store, options = {}) { slotFillRate: injectRuns > 0 ? injectedCount / (injectRuns * maxInjectSlots) : null }, topRecalled, + // #275 第五指标:归档净增速率 + 可压掉行数(口径见文件头) + archive: { + total: archivedTotal, + addedInWindow: archivedInWindow, + perDay: Math.round((archivedInWindow / windowDays) * 100) / 100, + compressible: { rows: compressibleRows, groups: compressibleGroups } + }, zombie: { activeCount, zombieCount, diff --git a/dsh-mneme/lib/service.js b/dsh-mneme/lib/service.js index 00302ecc..574458ab 100644 --- a/dsh-mneme/lib/service.js +++ b/dsh-mneme/lib/service.js @@ -2181,6 +2181,9 @@ export function createService({ store, mirror, config, onWrite, logger, document embedQuery, pushContentHistory, transaction, + // #275 拍板 5:升格吸收的 evidence 行随之归档,但 pinned 池(#249)永不自动归档 + // ——注册器不 import 这个集合,方向反了会成环,所以在这里注入。 + pinnedTypes: PINNED_MEMORY_TYPES, finalize: (rows) => { for (const row of rows) scheduleEmbed(row); } diff --git a/dsh-mneme/lib/tools.js b/dsh-mneme/lib/tools.js index 1f04a48a..0ef8c6bf 100644 --- a/dsh-mneme/lib/tools.js +++ b/dsh-mneme/lib/tools.js @@ -489,7 +489,11 @@ export function createTools(ctx, service, config, embedder) { "intersects evidence with real memory ids (all-fabricated evidence is rejected; unknown ids are dropped and the row " + "is tagged evidence_degraded), and dedupes: re-registering the same path or title supersedes the old row (the old " + "file is never touched; content_history stays traceable), while a merely near-duplicate summary of a different " + - "document row is rejected — update that row instead. Use for 'where is the conclusion doc for this project?' " + + "document row is rejected — update that row instead. On success the absorbed atomic evidence rows leave the active " + + "face in the same transaction (recoverable, nothing deleted); constraint/preference rows and other " + + "document/summary rows are never auto-archived, and keep_evidence_active: true opts out. Ids this same document " + + "absorbed on an earlier version still count as its evidence, so re-registering a new version is not read as " + + "fabricated evidence. Use for 'where is the conclusion doc for this project?' " + "lookups; atomic facts still go to memory_save.", parameters: { path: { type: "string", required: true, description: "Absolute path of the document file (~ is expanded); must already exist as a non-empty regular file. The full text stays agent-owned — this pipeline never touches it." }, @@ -497,7 +501,8 @@ export function createTools(ctx, service, config, embedder) { summary: { type: "string", required: true, description: "One-paragraph summary stored in the DB and used for injection (any language)" }, tags: { type: "array", items: { type: "string" }, description: "Optional tags (English recommended)" }, importance: { type: "integer", description: "1-5 (default 3); the summary row injects at the next-priority tier within documentInjectBudget when importance >= threshold" }, - evidence: { type: "array", items: { type: "string" }, description: "Memory ids this document is grounded in; each is verified against the store (fabricated evidence is rejected; unknown/archived ids are dropped and the row is tagged evidence_degraded)" }, + evidence: { type: "array", items: { type: "string" }, description: "Memory ids this document is grounded in (atomic facts, not other pointer rows); each is verified against the store — fabricated evidence is rejected, unknown/archived ids are dropped and the row is tagged evidence_degraded, except ids this same document absorbed on an earlier version, which stay its evidence. On success the absorbed atomic rows leave the active face unless keep_evidence_active is true." }, + keep_evidence_active: { type: "boolean", description: "Opt out of archiving the evidence rows absorbed by this document. Default false: after a successful registration the absorbed atomic rows leave the active face (recoverable, nothing deleted; constraint/preference rows and other document/summary rows are never auto-archived)." }, source: { type: "string", description: "Optional provenance" }, sensitivity: { type: "string", description: "Optional sensitivity label (free-form, e.g. personal). Part of the supersede matching key — same path/title with a different sensitivity stays a separate document." }, agent_scope: { type: "string", description: "Optional explicit agent-scope declaration (issue #170): 'global' or '*' makes this document visible to every agent; any other value narrows it to that label. Overrides the automatic carrier label for this write; honored even when automatic scope labeling is disabled." }, @@ -513,12 +518,13 @@ export function createTools(ctx, service, config, embedder) { superseded_id: { type: "string" }, evidence_kept: { type: "integer", required: true }, evidence_dropped: { type: "integer", required: true }, + evidence_archived: { type: "integer", required: true }, degraded: { type: "boolean", required: true } } }, render: (_args, value) => { const sup = value.superseded_id ? ` (supersedes ${value.superseded_id})` : ""; - const ev = ` | evidence: ${value.evidence_kept} kept, ${value.evidence_dropped} dropped${value.degraded ? " [degraded]" : ""}`; + const ev = ` | evidence: ${value.evidence_kept} kept, ${value.evidence_dropped} dropped, ${value.evidence_archived} archived${value.degraded ? " [degraded]" : ""}`; return TEXT_OUTPUT(`document ${value.action}: ${value.id}${sup}${ev}`); } }, @@ -557,13 +563,18 @@ export function createTools(ctx, service, config, embedder) { ...(args.sensitivity !== undefined ? { sensitivity: args.sensitivity } : {}), ...(agentLabel ? { agent_scope: agentLabel.value, agent_scope_source: agentLabel.source } : {}), ...(workspaceLabel ? { workspace_scope: workspaceLabel.value, workspace_scope_source: workspaceLabel.source } : {}) - }, { hiddenEvidenceIds: hiddenEvidence }); + }, { + hiddenEvidenceIds: hiddenEvidence, + // 默认开着归档(#275 拍板 5);agent 显式要保留活跃面时走 keep_evidence_active。 + archiveEvidence: args.keep_evidence_active !== true + }); return { action: result.action, id: result.memory.id, ...(result.superseded ? { superseded_id: result.superseded.id } : {}), evidence_kept: result.evidence_kept, evidence_dropped: result.evidence_dropped, + evidence_archived: result.evidence_archived, degraded: result.degraded }; } diff --git a/dsh-mneme/src/document.js b/dsh-mneme/src/document.js index f89d572a..79bf2d8c 100644 --- a/dsh-mneme/src/document.js +++ b/dsh-mneme/src/document.js @@ -60,20 +60,34 @@ export function isManagedDocumentPath(dir, path) { // id + 旧行 content_history 里的旧摘要(source=superseded)。 const supersededByPointer = (id) => `\n\n[superseded by ${id}]`; +// 升格吸收不碰的类型(#275 拍板 5 的边界):吸收的对象是「原子条」,而下面这两种 +// 行各有自己的生命周期,被归档会各自留下第二行—— +// document:本体是磁盘上的文件。归档它等于替用户撤下外档;而且 supersede 探测 +// 只看活跃行(store.list 默认排除归档),同一 path 下次注册会在旧行还挂着 +// 「新版本」语义时再铸一行,指针注记与 content_history 记账也走不到。 +// summary:dream 总览与叙述按 source 身份去重、并按注入档位常驻。归档它会让下一 +// 次做梦把它当不存在(候选集同样只取活跃行),于是库里多出一行同源总览。 +// 口径是保守方向的:拿不准的行一律不吸收——少做一步没有代价,收错一步是内容事故。 +const ABSORB_EXEMPT_TYPES = new Set(["document", "summary"]); + /** * Build the document registrar. Injected deps are service.js closure members * (store / embedQuery / transaction / finalize) so this module stays free of * service-internal wiring; `finalize` is the single write epilogue (mirror * sync + notify + re-embed) so callers never duplicate it. * - * @returns {(payload: object) => Promise} + * `pinnedTypes` 是注入的 #249 逐字保真池(service.js 的 PINNED_MEMORY_TYPES): + * 这里不 import 它在 service.js 里,是因为 service.js 反向 import 本模块—— + * 方向反了会成环。 + * + * @returns {(payload: object, opts?: object) => Promise} * `{ action: "created"|"superseded", memory, superseded?, evidence_kept, - * evidence_dropped, degraded }` + * evidence_dropped, evidence_archived, degraded }` * @throws flag 关 / 路径非法或文件缺失 / title 或 summary 为空 / evidence 全部 * 捏造 / 仅 vector 近重复(疑似重复外档,交 agent 裁决而不是静默二选一)。 */ -export function createDocumentRegistrar({ store, config, embedQuery, pushContentHistory, transaction, finalize }) { - return async function registerDocument(payload, { hiddenEvidenceIds = [] } = {}) { +export function createDocumentRegistrar({ store, config, embedQuery, pushContentHistory, transaction, finalize, pinnedTypes }) { + return async function registerDocument(payload, { hiddenEvidenceIds = [], archiveEvidence = true } = {}) { // opt-in 总闸(#230 对齐 #228 形态):关 = 注册入口整体不存在,错误信息 // 指回配置键,agent 能准确转告用户去开。 if (config?.documentMemoryEnabled !== true) { @@ -121,15 +135,14 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent const hidden = new Set((Array.isArray(hiddenEvidenceIds) ? hiddenEvidenceIds : []).map((id) => String(id))); const kept = []; const dropped = []; + // 存在但已归档的引用另记一份:它可能是被上一版文档升格吸收走的行(见下方 + // supersede 目标探测后的「认回」),不能一律按「不可用」处理。 + const archivedRefs = new Set(); for (const id of wanted) { const row = hidden.has(id) ? null : store.getById(id); + if (row?.archived) archivedRefs.add(id); (row && !row.archived ? kept : dropped).push(id); } - if (wanted.length > 0 && kept.length === 0) { - throw new Error( - `registerDocument: all ${wanted.length} evidence ids are unknown, archived or out of scope — fabricated evidence is rejected` - ); - } const title = String(payload?.title ?? "").trim(); if (!title) throw new Error("registerDocument: title is required"); @@ -162,6 +175,25 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent .filter((m) => scopeMatches(m)); const explicitTarget = scopeMatched.find((m) => samePath(m.doc_path, expanded)) ?? scopeMatched.find((m) => m.title.trim() === title); + // 升格吸收过的引用认回(#275 拍板 5 的配套):evidence 行在文档注册成功后就翻了 + // archived,于是「同一份文档出新版、evidence 照旧」这一最常规的路径会整批落在 + // dropped——那会被下面的捏造判据误报成「证据都是编的」。口径:已归档行仍可为 + // **吸收它的那份文档**背书(loser.evidence 就是它吸收走的名单),不为别的文档 + // 背书;捏造与跨 scope 照旧拒绝。 + if (explicitTarget) { + const own = new Set(explicitTarget.evidence ?? []); + for (const id of archivedRefs) { + if (!own.has(id)) continue; + const at = dropped.indexOf(id); + if (at >= 0) dropped.splice(at, 1); + kept.push(id); + } + } + if (wanted.length > 0 && kept.length === 0) { + throw new Error( + `registerDocument: all ${wanted.length} evidence ids are unknown, archived or out of scope — fabricated evidence is rejected` + ); + } if (!explicitTarget) { // vector 档:embedder 不可用/向量缺失一律跳过——去重是增强不是写入依赖 // (findSessionDuplicate 同原则)。probe 用 title+summary,与行向量 @@ -237,7 +269,25 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent store.setArchived(loser.id, true); superseded = store.getById(loser.id); } - return { created, superseded }; + // #275 拍板 5(#230 验收口径的补丁):升格吸收的 evidence 行随之退出活跃面—— + // 一次吸收 20 条,库里就是「21 行不是 1 行」。四条口径:①与本行同一事务(要么 + // 都成、要么都不成);②pinned(constraint / preference,即 #249 的逐字保真池) + // 永不自动归档——升格不能绕过它;③自有生命周期的类型(document / summary,见 + // ABSORB_EXEMPT_TYPES)不吸收,吸收的是原子条;④只翻标志位、内容与审计全留 + // (可恢复),opt-out 走 archiveEvidence。 + // pinnedTypes 拿不到就不做这一步:宁可少做,也不能把保真池当普通行收走。 + let evidenceArchived = 0; + if (archiveEvidence && pinnedTypes && typeof pinnedTypes.has === "function") { + for (const id of kept) { + // 事务内重读:期间被别的进程归档/遗忘的行不重复计数,也不误伤 pinned。 + const row = store.getById(id); + if (!row || row.archived || row.forgotten) continue; + if (pinnedTypes.has(row.type) || ABSORB_EXEMPT_TYPES.has(row.type)) continue; + store.setArchived(id, true); + evidenceArchived += 1; + } + } + return { created, superseded, evidenceArchived }; }); finalize(result.superseded ? [result.created, result.superseded] : [result.created]); return { @@ -246,6 +296,7 @@ export function createDocumentRegistrar({ store, config, embedQuery, pushContent ...(result.superseded ? { superseded: result.superseded } : {}), evidence_kept: kept.length, evidence_dropped: dropped.length, + evidence_archived: result.evidenceArchived, degraded: dropped.length > 0 }; }; diff --git a/dsh-mneme/src/recall-stats.js b/dsh-mneme/src/recall-stats.js index 70faaa86..8721b6e0 100644 --- a/dsh-mneme/src/recall-stats.js +++ b/dsh-mneme/src/recall-stats.js @@ -5,6 +5,8 @@ // 新内聚块从第一行就落在自己的文件里,service.js 保留 barrel 出口、调用方零改动。 // 纯读不写,绝不触发任何 write hook。 +import { scopeKeyOf } from "./scope.js"; + /** * 口径(issue #217 评论 2026-09-18,锚 5bd2dab): * - Top-N:窗口内 recall_runs.candidates(最终返回集)按 id 计数,join @@ -20,6 +22,17 @@ * - coverage:recallRecordDefault 开启前的窗口算不到,earliestRunAt 为 * null(窗口内无回执)或早于窗口起点时,前端标注可信度;扫描行数有 * 上限,超出标 truncated(degraded 口径),不静默少算。 + * - archive(#275 拍板的第五指标,与上面几项同位): + * · total = 归档区现有行数;addedInWindow / perDay = 窗口内进入归档区的行数 + * 与其日均(净增速率)。归档时刻只能取 updated_at(setArchived 翻标志位时刷 + * 它)——这是代理口径:对已归档行做一次 memory_update,或对同一行重复调 + * setArchived(…, true),都会被算成「本窗口新进归档」,perDay 因此偏高。精确 + * 口径要一个 archived_at 列(schema 变更,属第二批题材),本版不加;真正的 + * 「净」增本来也要跨快照比 total(物理删除落地后,差值才会由负向变化体现)。 + * · compressible = 可压掉行数:同 type、同 scope 三维且内容哈希完全相同的归档 + * 行里,多出来的那些(每组留一行)。用内容哈希而不是向量近重复,是因为回收 + * 动作本身会清掉归档行向量(clearArchivedEmbeddings)——指标不能建在它自己 + * 的输入会被回收掉的数据上;精确重复与 #254 的计量口径同一把尺。 * * @param {object} store - createStore 产物(只调用 listRecallRunsSince / all) * @param {{ windowDays?: number, exemptDays?: number }} [options] @@ -36,7 +49,8 @@ export function recallStats(store, options = {}) { const rawSlots = Number(options.maxInjectSlots ?? 5); const maxInjectSlots = Number.isInteger(rawSlots) && rawSlots >= 1 ? rawSlots : 5; const now = Date.now(); - const since = new Date(now - windowDays * 86400000).toISOString(); + const sinceMs = now - windowDays * 86400000; + const since = new Date(sinceMs).toISOString(); const { rows: runs, total } = store.listRecallRunsSince(since); const hits = new Map(); // id -> { count, title, source } @@ -67,8 +81,29 @@ export function recallStats(store, options = {}) { let activeCount = 0; let zombieCount = 0; let exemptCount = 0; + // 第五指标(#275):归档区一侧单独累计,分组键与 #254 的去重候选集同构 + // (type + 哈希 + scope 三维)——可压掉的必须是「本来就会被判成同一件事」的行。 + let archivedTotal = 0; + let archivedInWindow = 0; + const hashGroups = new Map(); for (const m of memories) { - if (m.archived || m.forgotten) continue; + if (m.archived) { + archivedTotal += 1; + const archivedMs = Date.parse(m.updated_at ?? ""); + if (Number.isFinite(archivedMs) && archivedMs >= sinceMs) archivedInWindow += 1; + if (m.content_hash) { + const key = [ + m.type ?? "", + m.content_hash, + scopeKeyOf(m.agent_scope) ?? "", + scopeKeyOf(m.workspace_scope) ?? "", + scopeKeyOf(m.sensitivity) ?? "" + ].join("\u0000"); + hashGroups.set(key, (hashGroups.get(key) ?? 0) + 1); + } + continue; + } + if (m.forgotten) continue; const createdMs = Date.parse(m.created_at ?? ""); // created_at 解析不了时无法证明已过机会期 → 归入豁免,宁漏勿误伤 if (!Number.isFinite(createdMs) || now - createdMs < EXEMPT_MS) { @@ -98,6 +133,14 @@ export function recallStats(store, options = {}) { .sort((a, b) => b.count - a.count || a.id.localeCompare(b.id)) .slice(0, 10); + let compressibleGroups = 0; + let compressibleRows = 0; + for (const n of hashGroups.values()) { + if (n < 2) continue; + compressibleGroups += 1; + compressibleRows += n - 1; + } + return { windowDays, generatedAt: new Date(now).toISOString(), @@ -113,6 +156,13 @@ export function recallStats(store, options = {}) { slotFillRate: injectRuns > 0 ? injectedCount / (injectRuns * maxInjectSlots) : null }, topRecalled, + // #275 第五指标:归档净增速率 + 可压掉行数(口径见文件头) + archive: { + total: archivedTotal, + addedInWindow: archivedInWindow, + perDay: Math.round((archivedInWindow / windowDays) * 100) / 100, + compressible: { rows: compressibleRows, groups: compressibleGroups } + }, zombie: { activeCount, zombieCount, diff --git a/dsh-mneme/src/service.js b/dsh-mneme/src/service.js index 00302ecc..574458ab 100644 --- a/dsh-mneme/src/service.js +++ b/dsh-mneme/src/service.js @@ -2181,6 +2181,9 @@ export function createService({ store, mirror, config, onWrite, logger, document embedQuery, pushContentHistory, transaction, + // #275 拍板 5:升格吸收的 evidence 行随之归档,但 pinned 池(#249)永不自动归档 + // ——注册器不 import 这个集合,方向反了会成环,所以在这里注入。 + pinnedTypes: PINNED_MEMORY_TYPES, finalize: (rows) => { for (const row of rows) scheduleEmbed(row); } diff --git a/dsh-mneme/src/tools.js b/dsh-mneme/src/tools.js index 1f04a48a..0ef8c6bf 100644 --- a/dsh-mneme/src/tools.js +++ b/dsh-mneme/src/tools.js @@ -489,7 +489,11 @@ export function createTools(ctx, service, config, embedder) { "intersects evidence with real memory ids (all-fabricated evidence is rejected; unknown ids are dropped and the row " + "is tagged evidence_degraded), and dedupes: re-registering the same path or title supersedes the old row (the old " + "file is never touched; content_history stays traceable), while a merely near-duplicate summary of a different " + - "document row is rejected — update that row instead. Use for 'where is the conclusion doc for this project?' " + + "document row is rejected — update that row instead. On success the absorbed atomic evidence rows leave the active " + + "face in the same transaction (recoverable, nothing deleted); constraint/preference rows and other " + + "document/summary rows are never auto-archived, and keep_evidence_active: true opts out. Ids this same document " + + "absorbed on an earlier version still count as its evidence, so re-registering a new version is not read as " + + "fabricated evidence. Use for 'where is the conclusion doc for this project?' " + "lookups; atomic facts still go to memory_save.", parameters: { path: { type: "string", required: true, description: "Absolute path of the document file (~ is expanded); must already exist as a non-empty regular file. The full text stays agent-owned — this pipeline never touches it." }, @@ -497,7 +501,8 @@ export function createTools(ctx, service, config, embedder) { summary: { type: "string", required: true, description: "One-paragraph summary stored in the DB and used for injection (any language)" }, tags: { type: "array", items: { type: "string" }, description: "Optional tags (English recommended)" }, importance: { type: "integer", description: "1-5 (default 3); the summary row injects at the next-priority tier within documentInjectBudget when importance >= threshold" }, - evidence: { type: "array", items: { type: "string" }, description: "Memory ids this document is grounded in; each is verified against the store (fabricated evidence is rejected; unknown/archived ids are dropped and the row is tagged evidence_degraded)" }, + evidence: { type: "array", items: { type: "string" }, description: "Memory ids this document is grounded in (atomic facts, not other pointer rows); each is verified against the store — fabricated evidence is rejected, unknown/archived ids are dropped and the row is tagged evidence_degraded, except ids this same document absorbed on an earlier version, which stay its evidence. On success the absorbed atomic rows leave the active face unless keep_evidence_active is true." }, + keep_evidence_active: { type: "boolean", description: "Opt out of archiving the evidence rows absorbed by this document. Default false: after a successful registration the absorbed atomic rows leave the active face (recoverable, nothing deleted; constraint/preference rows and other document/summary rows are never auto-archived)." }, source: { type: "string", description: "Optional provenance" }, sensitivity: { type: "string", description: "Optional sensitivity label (free-form, e.g. personal). Part of the supersede matching key — same path/title with a different sensitivity stays a separate document." }, agent_scope: { type: "string", description: "Optional explicit agent-scope declaration (issue #170): 'global' or '*' makes this document visible to every agent; any other value narrows it to that label. Overrides the automatic carrier label for this write; honored even when automatic scope labeling is disabled." }, @@ -513,12 +518,13 @@ export function createTools(ctx, service, config, embedder) { superseded_id: { type: "string" }, evidence_kept: { type: "integer", required: true }, evidence_dropped: { type: "integer", required: true }, + evidence_archived: { type: "integer", required: true }, degraded: { type: "boolean", required: true } } }, render: (_args, value) => { const sup = value.superseded_id ? ` (supersedes ${value.superseded_id})` : ""; - const ev = ` | evidence: ${value.evidence_kept} kept, ${value.evidence_dropped} dropped${value.degraded ? " [degraded]" : ""}`; + const ev = ` | evidence: ${value.evidence_kept} kept, ${value.evidence_dropped} dropped, ${value.evidence_archived} archived${value.degraded ? " [degraded]" : ""}`; return TEXT_OUTPUT(`document ${value.action}: ${value.id}${sup}${ev}`); } }, @@ -557,13 +563,18 @@ export function createTools(ctx, service, config, embedder) { ...(args.sensitivity !== undefined ? { sensitivity: args.sensitivity } : {}), ...(agentLabel ? { agent_scope: agentLabel.value, agent_scope_source: agentLabel.source } : {}), ...(workspaceLabel ? { workspace_scope: workspaceLabel.value, workspace_scope_source: workspaceLabel.source } : {}) - }, { hiddenEvidenceIds: hiddenEvidence }); + }, { + hiddenEvidenceIds: hiddenEvidence, + // 默认开着归档(#275 拍板 5);agent 显式要保留活跃面时走 keep_evidence_active。 + archiveEvidence: args.keep_evidence_active !== true + }); return { action: result.action, id: result.memory.id, ...(result.superseded ? { superseded_id: result.superseded.id } : {}), evidence_kept: result.evidence_kept, evidence_dropped: result.evidence_dropped, + evidence_archived: result.evidence_archived, degraded: result.degraded }; } diff --git a/dsh-mneme/test/client.test.js b/dsh-mneme/test/client.test.js index 48721da2..1cb6950b 100644 --- a/dsh-mneme/test/client.test.js +++ b/dsh-mneme/test/client.test.js @@ -25,6 +25,15 @@ test("client bundle is lib-only with no src counterpart", () => { assert.equal(existsSync(join(root, "lib/client.js")), true, "lib/client.js must exist"); }); +// 归档侧第五指标(#275)自己按「归档区为空就整段省略」门控,但整卡还有一道 hasData 门: +// 那道门只认召回回执与活跃僵尸行时,「全归档 + 窗口内无回执」的库会直接 return null, +// 指标永远不显示(自动评审 #312 指出的回归)。这里锁死 hasData 必须把 archive.total 计进来。 +test("status card: hasData counts the archive metric", () => { + const m = clientSource.match(/const hasData =[^;]+;/); + assert.ok(m, "the status card must declare hasData"); + assert.match(m[0], /d\.archive\?\.total/, "hasData must count archive.total, not just runs/active rows"); +}); + // The memory entry lives at the sidebar foot, not in the settings modal: the // migration must register into `sidebar.footer.action` (the list slot the // sidebar shell renders beside Settings) and must not keep a `settings.section` diff --git a/dsh-mneme/test/document.test.js b/dsh-mneme/test/document.test.js index 9b1fc55c..a5370559 100644 --- a/dsh-mneme/test/document.test.js +++ b/dsh-mneme/test/document.test.js @@ -8,7 +8,7 @@ import { mkdtemp, writeFile, rm } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import { createStore } from "../src/store.js"; -import { createService } from "../src/service.js"; +import { createService, PINNED_MEMORY_TYPES } from "../src/service.js"; import { createDocumentRegistrar } from "../src/document.js"; import { MEMORY_ITEM_SCHEMA } from "../src/tools.js"; import { TYPE_DECAY_DEFAULTS } from "../src/heat.js"; @@ -30,7 +30,7 @@ async function makeDocDir() { } /** 单元级 registrar:注入假 embedQuery,隔离 vector 档行为。 */ -function makeRegistrar(store, config = {}, embedQuery = async () => null) { +function makeRegistrar(store, config = {}, embedQuery = async () => null, { pinnedTypes = PINNED_MEMORY_TYPES } = {}) { const finalized = []; const register = createDocumentRegistrar({ store, @@ -42,7 +42,9 @@ function makeRegistrar(store, config = {}, embedQuery = async () => null) { ...(Array.isArray(existing?.content_history) ? existing.content_history : []) ].slice(0, 20), transaction: (fn) => fn(), - finalize: (rows) => finalized.push(...rows) + finalize: (rows) => finalized.push(...rows), + // #275 拍板 5:service.js 注入的 #249 逐字保真池(这里同款注入,保持行为一致) + pinnedTypes }); return { register, finalized }; } @@ -147,6 +149,152 @@ test("evidence intersection: valid subset kept, unknown dropped + degraded tag, } }); +// ============================ 升格吸收的 evidence 归档(#275 拍板 5,同事务) + +// 拍板原文:registerDocument 成功后同一事务把 absorbed 的 evidence 行翻 archived +// (可恢复、审计全留),带 opt-out;pinned(constraint / preference)永不自动归档; +// supersede 的 loser 行同样翻 archived(后者本来就是本模块既有行为,这里一并锁住)。 + +test("absorbed evidence rows are archived with the document; pinned rows are exempt", async () => { + const { store, service, close } = makeService({ documentMemoryEnabled: true }); + const { writeDoc, cleanup } = await makeDocDir(); + try { + const pinned = service.saveWithDedupe({ type: "constraint", title: "C", content: "边界条件" }).memory; + const decision = service.saveWithDedupe({ type: "decision", title: "D", content: "一个决定" }).memory; + const p = await writeDoc("absorb.md"); + + const res = await service.registerDocument({ + path: p, title: "Absorb", summary: "summary", evidence: [pinned.id, decision.id] + }); + + assert.equal(res.evidence_kept, 2, "引用本身照旧全留下:doc 行的 evidence 数组是正向链"); + assert.equal(res.evidence_archived, 1, "只有非 pinned 的那条被动"); + const after = service.getById(decision.id); + assert.equal(after.archived, true, "被吸收的行退出活跃面(「21 行不是 1 行」)"); + assert.equal(after.content, "一个决定", "只翻标志位:内容与审计全留(可恢复)"); + assert.equal(service.getById(pinned.id).archived, false, "#249 的逐字保真池,升格不能绕过"); + assert.equal(store.list().some((m) => m.id === decision.id), false, "默认面不再列出吸收行"); + } finally { + cleanup(); + close(); + } +}); + +test("archiveEvidence: false(工具侧 keep_evidence_active)保留吸收行的活跃面", async () => { + const { service, close } = makeService({ documentMemoryEnabled: true }); + const { writeDoc, cleanup } = await makeDocDir(); + try { + const d = service.saveWithDedupe({ type: "decision", title: "D", content: "c" }).memory; + const res = await service.registerDocument( + { path: await writeDoc("optout.md"), title: "OptOut", summary: "s", evidence: [d.id] }, + { archiveEvidence: false } + ); + assert.equal(res.evidence_kept, 1); + assert.equal(res.evidence_archived, 0, "opt-out 时不归档"); + assert.equal(service.getById(d.id).archived, false); + } finally { + cleanup(); + close(); + } +}); + +test("re-registering a document whose evidence was absorbed keeps the references", async () => { + // 这条防的是「补丁把常规路径打坏」:吸收行一旦归档,重注册同一份文档(出新版) + // 时同一批 evidence 会全落 dropped,进而撞上捏造判据——最常规的路径反而报错。 + const { service, close } = makeService({ documentMemoryEnabled: true }); + const { writeDoc, cleanup } = await makeDocDir(); + try { + const d = service.saveWithDedupe({ type: "decision", title: "D", content: "c" }).memory; + const p = await writeDoc("ver.md"); + const first = await service.registerDocument({ path: p, title: "Ver", summary: "v1", evidence: [d.id] }); + assert.equal(first.evidence_archived, 1); + + const second = await service.registerDocument({ path: p, title: "Ver", summary: "v2", evidence: [d.id] }); + assert.equal(second.action, "superseded"); + assert.deepEqual(second.memory.evidence, [d.id], "吸收行仍为它所属的那份文档背书"); + assert.equal(second.evidence_kept, 1); + assert.equal(second.degraded, false, "不误报成捏造证据"); + // 认回是窄口径的:已归档行只为**吸收它的那份**文档背书。别的文档(新注册、无 + // 同名目标)拿它当证据,照旧走既有的捏造判据——不因为这条补丁放宽。 + const other = await writeDoc("other.md"); + await assert.rejects( + () => service.registerDocument({ path: other, title: "Other", summary: "s", evidence: [d.id] }), + /fabricated evidence is rejected/ + ); + } finally { + cleanup(); + close(); + } +}); + +test("registrar without the pinned pool wiring archives nothing (fail-safe)", async () => { + // 拿不到 pinned 集合就不做这一步:宁可少做,也不能把保真池当普通行收走。 + const store = createStore(":memory:"); + const { writeDoc, cleanup } = await makeDocDir(); + try { + const d = store.save({ type: "decision", title: "D", content: "c" }); + const { register } = makeRegistrar(store, { documentMemoryEnabled: true }, async () => null, { pinnedTypes: null }); + const res = await register({ path: await writeDoc("noset.md"), title: "NoSet", summary: "s", evidence: [d.id] }); + assert.equal(res.evidence_archived, 0); + assert.equal(store.getById(d.id).archived, false); + } finally { + cleanup(); + store.close(); + } +}); + +test("自有生命周期的类型不被吸收:document 行保持活跃,同 path 的 supersede 链不断", async () => { + // 回归点(单盲审查 H1):把另一份 document 行当 evidence 时,若照普通行吸收,它会 + // 被静默归档——走不到 loser 分支(没有 [superseded by] 指针与 content_history), + // 而 supersede 探测只看活跃行,于是重注册同 path 会再铸一行、同一个文件留下两行。 + const { store, service, close } = makeService({ documentMemoryEnabled: true }); + const { writeDoc, cleanup } = await makeDocDir(); + try { + const pA = await writeDoc("a.md"); + const docA = (await service.registerDocument({ path: pA, title: "DocA", summary: "v1" })).memory; + const docB = await service.registerDocument({ + path: await writeDoc("b.md"), title: "DocB", summary: "grounded in A", evidence: [docA.id] + }); + assert.equal(docB.evidence_archived, 0, "document 行不参与吸收"); + assert.equal(service.getById(docA.id).archived, false, "被引用不改变它的活跃状态"); + + const again = await service.registerDocument({ path: pA, title: "DocA", summary: "v2" }); + assert.equal(again.action, "superseded", "指针行仍能被 supersede 探测看到"); + assert.equal(again.superseded.id, docA.id); + assert.ok(again.superseded.content.includes(again.memory.id), "旧行拿到 [superseded by] 指针"); + assert.equal(store.all().filter((m) => m.doc_path === pA).length, 2, "同一 path 一行新版 + 一行带指针的旧版,没有多余的第三行"); + } finally { + cleanup(); + close(); + } +}); + +test("自有生命周期的类型不被吸收:summary(dream 总览)保持活跃,不会被当成不存在重铸", async () => { + // 回归点:summary 按 source 身份去重(dream 总览)且按注入档位常驻。被吸收归档后 + // 去重候选集(store.list 默认排除归档)看不到它 → 下一次做梦会再铸一行同源总览。 + const { service, close } = makeService({ documentMemoryEnabled: true }); + const { writeDoc, cleanup } = await makeDocDir(); + try { + const digest = service.saveWithDedupe({ + type: "summary", title: "当前项目状态", content: "resident digest", source: "dream", importance: 5 + }).memory; + const res = await service.registerDocument({ + path: await writeDoc("with-digest.md"), title: "DocW", summary: "s", evidence: [digest.id] + }); + assert.equal(res.evidence_archived, 0, "summary 不参与吸收"); + assert.equal(service.getById(digest.id).archived, false); + + const again = service.saveWithDedupe({ + type: "summary", title: "当前项目状态", content: "resident digest v2", source: "dream", importance: 5 + }); + assert.equal(again.action, "merged", "下一次做梦仍认得出这行,不会铸出第二行同源总览"); + assert.equal(again.memory.id, digest.id); + } finally { + cleanup(); + close(); + } +}); + // ============================================================ supersede 记账(DoD 2) test("re-registering the same path supersedes: old row archived + traceable, file untouched", async () => { diff --git a/dsh-mneme/test/recall-stats.test.js b/dsh-mneme/test/recall-stats.test.js index ae862511..ef7bcdcd 100644 --- a/dsh-mneme/test/recall-stats.test.js +++ b/dsh-mneme/test/recall-stats.test.js @@ -121,3 +121,69 @@ test("windowDays/exemptDays 越界回默认;扫描超上限标 truncated", () assert.equal(total, 2, "窗口总数不截断 → truncated = total > rows.length"); store.close(); }); + +// ============================ 第五指标(#275):归档净增速率 + 可压掉行数 + +test("归档净增速率:按 updated_at 取归档时刻,窗外归档只进总数不进净增", () => { + const { store, service } = setup(); + const fresh = service.saveWithDedupe({ type: "history", title: "刚归档", content: "x", importance: 3 }).memory; + const old = service.saveWithDedupe({ type: "history", title: "早归档", content: "y", importance: 3 }).memory; + service.saveWithDedupe({ type: "history", title: "活跃件", content: "z", importance: 3 }); + store.setArchived(fresh.id, true); + store.setArchived(old.id, true); + // 归档时刻 = 最后一次写(setArchived 刷 updated_at):把这行推到窗口外 + store.db.prepare("UPDATE memories SET updated_at = ? WHERE id = ?") + .run(new Date(Date.now() - 100 * 86400000).toISOString(), old.id); + + const stats = service.recallStats({ windowDays: 30, exemptDays: 0 }); + assert.equal(stats.archive.total, 2, "归档区现有行数(活跃件不算)"); + assert.equal(stats.archive.addedInWindow, 1, "窗外那次归档不算本窗口净增"); + assert.equal(stats.archive.perDay, Math.round((1 / 30) * 100) / 100); + store.close(); +}); + +test("可压掉行数:同 type 同 scope 的内容哈希精确重复(跨 scope / 活跃行都不算)", () => { + const { store, service } = setup(); + // 标题只差大小写、正文只差标点与空白 → 归一化后同哈希(#254 的口径) + const a = store.save({ type: "history", title: "Task A", content: "line1.\n\nline2" }); + const b = store.save({ type: "history", title: "task a", content: "line1 line2" }); + const otherScope = store.save({ type: "history", title: "Task A", content: "line1 line2", agent_scope: "other" }); + store.save({ type: "history", title: "Task A", content: "line1 line2" }); + for (const id of [a.id, b.id, otherScope.id]) store.setArchived(id, true); + + const stats = service.recallStats({ windowDays: 30, exemptDays: 0 }); + assert.equal(stats.archive.compressible.groups, 1, "跨 scope 不互判(与去重候选集同一把尺)"); + assert.equal(stats.archive.compressible.rows, 1, "每组留一行,多出来的才叫可压掉"); + assert.equal(stats.archive.total, 3, "跨 scope 行照进总数,只是不参与可压掉"); + assert.equal(stats.archive.addedInWindow, 3, "三行都是本窗口内归档的"); + store.close(); +}); + +test("可压掉行数:向量口径不可用时的兜底——指标不依赖 embedding", () => { + // 回归点:回收动作会清掉归档行向量(clearArchivedEmbeddings),若指标建在向量近 + // 重复上,回收一跑它就归零。这里清完向量后指标必须不变。 + const { store, service } = setup(); + const a = store.save({ type: "history", title: "同题", content: "同一件事" }); + const b = store.save({ type: "history", title: "同题", content: "同一件事" }); + store.setArchived(a.id, true); + store.setArchived(b.id, true); + assert.equal(service.recallStats({ windowDays: 30, exemptDays: 0 }).archive.compressible.rows, 1); + + store.clearArchivedEmbeddings(); + assert.equal( + service.recallStats({ windowDays: 30, exemptDays: 0 }).archive.compressible.rows, + 1, + "清向量不影响可压掉行数" + ); + store.close(); +}); + +test("没有归档行时给 0,不返回 null / NaN", () => { + const { store, service } = setup(); + service.saveWithDedupe({ type: "project", title: "只有活跃", content: "x", importance: 3 }); + const stats = service.recallStats({ windowDays: 30, exemptDays: 0 }); + assert.deepEqual(stats.archive, { + total: 0, addedInWindow: 0, perDay: 0, compressible: { rows: 0, groups: 0 } + }); + store.close(); +});