From 270990b6a1c89d30b48c4da5f2ff1633d3ac72f5 Mon Sep 17 00:00:00 2001 From: MXAntian Date: Wed, 23 Sep 2026 11:25:26 +0800 Subject: [PATCH 1/2] fix(v2.11.1): stop the recall feedback loop; recall on what the user typed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four fixes found auditing a long-lived store (10k rows, 14 days of hook logs). 1. recallForClients bumped access for its whole candidate pool. It over-fetches (up to 30 rows when filtering) and trims to the caller's limit, but never passed _deferAccessBump — so every hook call added +1 to ~30 rows while returning 3. buildMemoryContext got this fix in #7; this path was missed. Now only returned rows are bumped. 2. freqScore weight 0.10 -> 0.02. It saturates at 20 accesses, so on a long-lived store it is effectively binary, and 0.10 is ~38 rank positions at RRF's ~0.0026 spacing. With (1) inflating the counts, the loop fed itself: one row reached 46% of two weeks of hook recalls. At 0.02 it is a tiebreak (~8 positions max). 3. The DB path is resolved when the store opens, not at import. A host that imports mneme and then loads its own env sets TOKENMEM_DB_PATH after ES import hoisting has evaluated the module-level const, so the store silently opened the fallback DB. With several tenants on one machine, one tenant's writes landed in another's store. 4. prompt-recall hook: recall on user-authored text only; default to meta+semi; require vector evidence before the trim. - UserPromptSubmit also delivers , and prepended blocks. 631 of 688 fast-path queries were that wrapper text. New userPromptText() strips it. - The triggers are operational (paths, ports, restarts), and the write-time meta gate downgrades anything carrying those to semi_abstract — so a meta-only default could never match the answers the hook is for. - Ask for require_vec first so the few slots are not taken by FTS/entity matches on a shared common noun; retry without it on zero rows so zero-config installs keep plain FTS + the consensus gate. When vectors are present, gate on MNEME_MAX_VEC_DISTANCE (0.95) instead of hit count. Tests: access-bump-scope and db-path-late-env integration tests (both red on main: 12 rows bumped for 2 returned; store opened the fallback path), plus userPromptText cases. Wired into CI. Existing ranking / endpoint / contract / cold-pool / anchor / hygiene / embedding-timeout / meta-gate / hooks suites green. Co-authored-by: 千夏 Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ci.yml | 10 +++++ access-bump-scope.integration.test.mjs | 55 ++++++++++++++++++++++++++ db-path-late-env.integration.test.mjs | 47 ++++++++++++++++++++++ docs/configuring-your-agent.md | 5 ++- docs/configuring-your-agent.zh-CN.md | 5 ++- hooks/prompt-recall-trigger.mjs | 18 +++++++++ hooks/prompt-recall-trigger.test.mjs | 17 ++++++++ hooks/prompt-recall.mjs | 39 ++++++++++++++---- index.mjs | 35 ++++++++++++++-- level-rank-offset.test.mjs | 2 +- package.json | 2 +- 11 files changed, 218 insertions(+), 17 deletions(-) create mode 100644 access-bump-scope.integration.test.mjs create mode 100644 db-path-late-env.integration.test.mjs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b4a6389..e1494f2 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -66,6 +66,16 @@ jobs: TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-ranking-importance.db run: node ranking-importance.integration.test.mjs + - name: Integration test (access bump — only returned rows, not the candidate pool) + env: + TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-access-bump.db + run: node access-bump-scope.integration.test.mjs + + - name: Integration test (DB path read at open time, not import time) + env: + TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-late-db.db + run: node db-path-late-env.integration.test.mjs + - name: Integration test (encoding-damage detection, both write paths) env: TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-encoding-damage.db diff --git a/access-bump-scope.integration.test.mjs b/access-bump-scope.integration.test.mjs new file mode 100644 index 0000000..b6ac1d3 --- /dev/null +++ b/access-bump-scope.integration.test.mjs @@ -0,0 +1,55 @@ +// recallForClients over-fetches a candidate pool (up to 30 rows when filtering) +// and trims it to the caller's limit. Access counts must move only for the rows +// that were returned. When the whole pool was bumped, every hook call added +1 to +// ~30 rows while showing 3, and since access frequency feeds ranking and decay the +// rows that were already on top stayed there: on one 10k-row store a single row +// reached 46% of two weeks of hook recalls. +// +// Run: TOKENMEM_DB_PATH=/tmp/x.db node access-bump-scope.integration.test.mjs +import { initMemory, closeMemory, storeMemory, recallForClients } from './index.mjs' +import Database from 'better-sqlite3' + +const DB_PATH = process.env.TOKENMEM_DB_PATH +if (!DB_PATH) { console.error('FATAL: set TOKENMEM_DB_PATH'); process.exit(2) } + +let pass = 0, fail = 0 +const check = (label, cond, detail = '') => { + if (cond) { pass++; console.log(`✓ ${label}`) } + else { fail++; console.log(`✗ ${label}${detail ? ' — ' + detail : ''}`) } +} + +initMemory() +const marker = `zzbump${Math.floor(Math.random() * 1e6)}` +const ids = [] +for (let i = 0; i < 12; i++) { + ids.push(storeMemory({ + content: `${marker} pool row ${i} ${'x'.repeat(i)}`, + importance: 7, memoryLevel: 'semi_abstract', memoryType: 'long_term', + })) +} + +const readCounts = () => { + const db = new Database(DB_PATH, { readonly: true }) + const rows = db.prepare(`SELECT rowid, access_count FROM memories WHERE rowid IN (${ids.map(() => '?').join(',')})`).all(...ids) + db.close() + return new Map(rows.map(r => [String(r.rowid), r.access_count || 0])) +} + +const before = readCounts() +// min_importance > 0 is what makes recallForClients over-fetch — the shape the hooks send. +const res = await recallForClients({ query: marker, limit: 2, minImportance: 1, source: 'test' }) +const after = readCounts() + +const returned = new Set(res.hits.map(h => String(h.id))) +const bumped = [...after].filter(([id, n]) => n > (before.get(id) || 0)).map(([id]) => id) + +check('the pool held more candidates than were returned', res.hits.length === 2 && bumped.length <= 2, + `returned=${res.hits.length} bumped=${bumped.length}`) +check('every returned row was bumped', [...returned].every(id => bumped.includes(id)), + `returned=${[...returned]} bumped=${bumped}`) +check('no row outside the result was bumped', bumped.every(id => returned.has(id)), + `bumped=${bumped} returned=${[...returned]}`) + +closeMemory() +console.log(`\n${fail === 0 ? 'PASS' : 'FAIL'}: ${pass} passed / ${fail} failed`) +process.exit(fail === 0 ? 0 : 1) diff --git a/db-path-late-env.integration.test.mjs b/db-path-late-env.integration.test.mjs new file mode 100644 index 0000000..f8f782e --- /dev/null +++ b/db-path-late-env.integration.test.mjs @@ -0,0 +1,47 @@ +// The DB path is read when the store opens, not when the module is imported. +// +// A host that imports mneme and then loads its own env file sets +// TOKENMEM_DB_PATH after ES import hoisting has already run this module's top +// level. With a module-level const the store silently opened the fallback DB +// beside index.mjs — and with more than one tenant on a machine, one tenant's +// writes landed in another's store. +// +// This file re-runs itself as a child with TOKENMEM_DB_PATH unset at import +// time, then sets it in the body — the host's order — and checks where the +// store actually opened. +// +// Run: TOKENMEM_DB_PATH=/tmp/x.db node db-path-late-env.integration.test.mjs +import { spawnSync } from 'node:child_process' +import { existsSync, unlinkSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +if (process.env.MNEME_LATE_DB_CHILD) { + const { initMemory, closeMemory, storeMemory } = await import('./index.mjs') + process.env.TOKENMEM_DB_PATH = process.env.MNEME_LATE_DB_CHILD // host sets it after import + initMemory() + storeMemory({ content: 'late env marker', importance: 5, memoryLevel: 'semi_abstract', memoryType: 'long_term' }) + closeMemory() + process.exit(0) +} + +const target = process.env.TOKENMEM_DB_PATH +if (!target) { console.error('FATAL: set TOKENMEM_DB_PATH'); process.exit(2) } +const latePath = target + '.late.db' +for (const sfx of ['', '-shm', '-wal']) { if (existsSync(latePath + sfx)) unlinkSync(latePath + sfx) } + +const env = { ...process.env, MNEME_LATE_DB_CHILD: latePath } +delete env.TOKENMEM_DB_PATH +delete env.MNEME_DB_PATH +const r = spawnSync(process.execPath, [fileURLToPath(import.meta.url)], { env, encoding: 'utf-8', timeout: 30000 }) + +let pass = 0, fail = 0 +const check = (label, cond, detail = '') => { + if (cond) { pass++; console.log(`✓ ${label}`) } + else { fail++; console.log(`✗ ${label}${detail ? ' — ' + detail : ''}`) } +} +check('child ran cleanly', r.status === 0, (r.stderr || '').slice(0, 300)) +check('store opened the path set after import', existsSync(latePath), latePath) + +for (const sfx of ['', '-shm', '-wal']) { try { unlinkSync(latePath + sfx) } catch {} } +console.log(`\n${fail === 0 ? 'PASS' : 'FAIL'}: ${pass} passed / ${fail} failed`) +process.exit(fail === 0 ? 0 : 1) diff --git a/docs/configuring-your-agent.md b/docs/configuring-your-agent.md index 18c5c80..ff6c2f8 100644 --- a/docs/configuring-your-agent.md +++ b/docs/configuring-your-agent.md @@ -234,9 +234,10 @@ be re-injected within one Claude Code session. | `MNEME_DB_PATH` | mneme's own `engram.db` | Alias for `TOKENMEM_DB_PATH`; picks the DB file | | `MNEME_INDEX_PATH` | `/index.mjs` | Override the engine entry point | | `MNEME_MIN_IMPORTANCE` | `6` | Floor for prompt-recall hits | -| `MNEME_LEVEL` | `meta_knowledge` | Prompt-recall level filter | +| `MNEME_LEVEL` | `meta_knowledge,semi_abstract` | Prompt-recall level filter | +| `MNEME_MAX_VEC_DISTANCE` | `0.95` | Prompt-recall drops hits farther than this when vectors are available | | `MNEME_LIMIT` | `5` | Prompt-recall candidate cap | -| `MNEME_MIN_CONSENSUS` | `2` | Prompt-recall skip if fewer hits | +| `MNEME_MIN_CONSENSUS` | `2` | Prompt-recall skip if fewer hits (only when no vector evidence) | | `MNEME_TOOL_MIN_IMPORTANCE` | `6` | Floor for tool-recall hits | | `MNEME_TOOL_LEVEL` | `meta_knowledge,semi_abstract` | Tool-recall level filter | | `MNEME_TOOL_LIMIT` | `4` | Tool-recall candidate cap | diff --git a/docs/configuring-your-agent.zh-CN.md b/docs/configuring-your-agent.zh-CN.md index b1b5eb8..d7d1ba0 100644 --- a/docs/configuring-your-agent.zh-CN.md +++ b/docs/configuring-your-agent.zh-CN.md @@ -206,9 +206,10 @@ MCP tool 是 pull 模式——Agent 自己决定何时 `recall_memory`。有些 | `MNEME_DB_PATH` | mneme 自带 `engram.db` | `TOKENMEM_DB_PATH` 的别名,选 DB 文件 | | `MNEME_INDEX_PATH` | `/index.mjs` | 覆盖引擎入口 | | `MNEME_MIN_IMPORTANCE` | `6` | prompt-recall 命中门槛 | -| `MNEME_LEVEL` | `meta_knowledge` | prompt-recall level 过滤 | +| `MNEME_LEVEL` | `meta_knowledge,semi_abstract` | prompt-recall level 过滤 | +| `MNEME_MAX_VEC_DISTANCE` | `0.95` | 有向量时,prompt-recall 丢弃距离大于此值的命中 | | `MNEME_LIMIT` | `5` | prompt-recall 候选上限 | -| `MNEME_MIN_CONSENSUS` | `2` | prompt-recall 少于此数就不注入 | +| `MNEME_MIN_CONSENSUS` | `2` | prompt-recall 少于此数就不注入(仅在没有向量证据时生效) | | `MNEME_TOOL_MIN_IMPORTANCE` | `6` | tool-recall 命中门槛 | | `MNEME_TOOL_LEVEL` | `meta_knowledge,semi_abstract` | tool-recall level 过滤 | | `MNEME_TOOL_LIMIT` | `4` | tool-recall 候选上限 | diff --git a/hooks/prompt-recall-trigger.mjs b/hooks/prompt-recall-trigger.mjs index 33e456a..6cb85ed 100644 --- a/hooks/prompt-recall-trigger.mjs +++ b/hooks/prompt-recall-trigger.mjs @@ -22,6 +22,24 @@ const TRIGGERS = [ /(?:路径|目录|位置|端口|凭证|密钥|环境变量|配置|脚本|工具|命令|启动|重启|守护进程)/, ] +// UserPromptSubmit carries more than what the user typed. In Claude Code, +// background-agent reports arrive as , messages from other +// sessions as , and app notices get prepended as +// . Recalling on the raw prompt means recalling on that +// wrapper text: on one store, 631 of 688 fast-path queries over two weeks were +// wrapper text, and those reports are dense with exactly the infrastructure +// nouns the triggers below look for. +// +// Returns '' when nothing user-authored remains. +const NOT_USER_INPUT = /^<(task-notification|cross-session-message)\b/ + +export function userPromptText(raw) { + if (!raw || typeof raw !== 'string') return '' + const text = raw.replace(/[\s\S]*?<\/system-reminder>/g, '').trim() + if (NOT_USER_INPUT.test(text)) return '' + return text +} + export function shouldTriggerPromptRecall(prompt) { if (!prompt || prompt.length < 4) return false if (prompt.length > 1500) return false diff --git a/hooks/prompt-recall-trigger.test.mjs b/hooks/prompt-recall-trigger.test.mjs index 7252d53..6a9189f 100644 --- a/hooks/prompt-recall-trigger.test.mjs +++ b/hooks/prompt-recall-trigger.test.mjs @@ -16,3 +16,20 @@ test('short steering and long pasted text stay outside the gate', () => { assert.equal(shouldTriggerPromptRecall('推进'), false) assert.equal(shouldTriggerPromptRecall('召回'.repeat(751)), false) }) + +test('only user-authored text reaches the trigger and the query', async () => { + const { userPromptText } = await import('./prompt-recall-trigger.mjs') + // App notice prepended to a real prompt: the notice goes, the prompt stays. + assert.equal( + userPromptText('\nThe user started background task X ("fix daemon restart")\n\nwhy is recall noisy'), + 'why is recall noisy', + ) + // Background-agent reports and other sessions' messages are not user input, + // even though they are full of the nouns the triggers look for. + assert.equal(userPromptText('\ndaemon config path port restart\n'), '') + assert.equal(userPromptText('where is the config'), '') + assert.equal(shouldTriggerPromptRecall(userPromptText('how to restart the daemon')), false) + // Plain prompts pass through untouched. + assert.equal(userPromptText(' how to restart the daemon '), 'how to restart the daemon') + assert.equal(userPromptText(undefined), '') +}) diff --git a/hooks/prompt-recall.mjs b/hooks/prompt-recall.mjs index 64bd80f..9e7b33f 100644 --- a/hooks/prompt-recall.mjs +++ b/hooks/prompt-recall.mjs @@ -25,7 +25,9 @@ // MNEME_DB_PATH alias for TOKENMEM_DB_PATH; where mneme's engram.db lives // MNEME_INDEX_PATH override for index.mjs (default: ../index.mjs) // MNEME_MIN_IMPORTANCE floor for hits (default: 6) -// MNEME_LEVEL recall level filter (default: meta_knowledge) +// MNEME_LEVEL recall level filter (default: meta_knowledge,semi_abstract) +// MNEME_MAX_VEC_DISTANCE drop hits farther than this when the server returned +// vector evidence (default: 0.95; ignored without embeddings) // MNEME_LIMIT max recall candidates (default: 5) // MNEME_MIN_CONSENSUS hide injection if hits < this (default: 2) // MNEME_STATE_DIR session-dedup file dir (default: ~/.claude/hooks) @@ -43,7 +45,7 @@ import { spawnSync } from 'node:child_process' import { readFileSync, writeFileSync, existsSync, mkdirSync, renameSync } from 'node:fs' import { resolve, dirname } from 'node:path' import { fileURLToPath } from 'node:url' -import { shouldTriggerPromptRecall } from './prompt-recall-trigger.mjs' +import { shouldTriggerPromptRecall, userPromptText } from './prompt-recall-trigger.mjs' const __dirname = dirname(fileURLToPath(import.meta.url)) const HOME = process.env.USERPROFILE || process.env.HOME || __dirname @@ -98,7 +100,12 @@ const CFG = { httpTimeoutMs: intEnv('MNEME_HTTP_TIMEOUT_MS', 1500), indexPath: process.env.MNEME_INDEX_PATH || resolve(__dirname, '..', 'index.mjs'), minImportance: intEnv('MNEME_MIN_IMPORTANCE', 6), - level: process.env.MNEME_LEVEL || 'meta_knowledge', + // semi_abstract is in the default on purpose. The triggers are operational + // (paths, ports, restarts, config), and the write-time meta gate downgrades + // anything carrying a path, port or version to semi_abstract — so under a + // meta-only filter the answers this hook exists to find could never match. + level: process.env.MNEME_LEVEL || 'meta_knowledge,semi_abstract', + maxVecDistance: (() => { const n = Number(process.env.MNEME_MAX_VEC_DISTANCE); return Number.isFinite(n) && n > 0 ? n : 0.95 })(), limit: intEnv('MNEME_LIMIT', 5), minConsensus: intEnv('MNEME_MIN_CONSENSUS', 2), stateDir: process.env.MNEME_STATE_DIR || resolve(HOME, '.claude', 'hooks'), @@ -161,11 +168,16 @@ async function recallOverHttp(body) { } catch { return null } } async function runRecall(query, sessionId) { - const viaHttp = await recallOverHttp({ + const httpReq = (requireVec) => recallOverHttp({ query: query, limit: CFG.limit, min_importance: CFG.minImportance, level: CFG.level, + // Filter to rows with vector evidence BEFORE the server trims to `limit`. + // Otherwise the few slots go to character-level FTS and entity matches — + // on CJK text that is mostly generic rows sharing one common noun — and the + // semantically relevant rows sit just below the cut. + require_vec: requireVec, source: 'mneme-prompt-recall', session_id: sessionId, // Tell the server how long we will actually wait, so a slow embedding @@ -175,6 +187,11 @@ async function runRecall(query, sessionId) { // so the embedding gets httpTimeoutMs − 250 — 1250 ms at the default. deadline_ms: Math.max(300, CFG.httpTimeoutMs - 100), }) + // Zero rows under require_vec means no embeddings configured, the embedding + // call degraded, or genuinely nothing — ask again without it so zero-config + // installs keep the plain FTS behaviour (and the consensus gate below). + let viaHttp = await httpReq(true) + if (viaHttp && viaHttp.hits.length === 0) viaHttp = await httpReq(false) if (viaHttp) return viaHttp const args = [ @@ -219,7 +236,7 @@ process.stdin.on('end', async () => { try { payload = JSON.parse(input || '{}') } catch { process.exit(0) } const sessionId = payload.session_id || payload.sessionId || 'unknown' - const prompt = (payload.prompt || '').trim() + const prompt = userPromptText(payload.prompt || '') if (!shouldTriggerPromptRecall(prompt)) process.exit(0) @@ -227,8 +244,16 @@ process.stdin.on('end', async () => { const recalled = await runRecall(query, sessionId) if (!recalled || !Array.isArray(recalled.hits)) process.exit(0) - const hits = recalled.hits - if (hits.length < CFG.minConsensus) process.exit(0) + // With vector evidence present, relevance is measurable — gate on it and + // skip the count heuristic. Without it (no embeddings), fall back to + // requiring agreement between several FTS hits. + let hits = recalled.hits + if (hits.some(h => typeof h.vec_distance === 'number')) { + hits = hits.filter(h => typeof h.vec_distance === 'number' && h.vec_distance <= CFG.maxVecDistance) + if (hits.length === 0) process.exit(0) + } else if (hits.length < CFG.minConsensus) { + process.exit(0) + } const injected = loadInjected(sessionId) const fresh = hits.filter(h => !injected.has(h.id)) diff --git a/index.mjs b/index.mjs index b0937bc..4b1b35d 100644 --- a/index.mjs +++ b/index.mjs @@ -64,7 +64,14 @@ try { // the pre-rename era — silent migration would create a split-brain second // DB; we'd rather keep using the populated one) // 3. tokenmem.db (default for fresh installs) -const DB_PATH = process.env.TOKENMEM_DB_PATH +// +// Resolved at call time, not import time. A host that imports this module and +// then loads its own env file sets TOKENMEM_DB_PATH too late for a module-level +// const — ES imports are hoisted above the host's body — so the store silently +// opened the fallback DB. With more than one tenant per machine that means +// writes land in the wrong store. getDb() first runs inside initMemory(), after +// the host's env is ready. +const resolveDbPath = () => process.env.TOKENMEM_DB_PATH || (existsSync(resolve(__dirname, 'engram.db')) ? resolve(__dirname, 'engram.db') : resolve(__dirname, 'tokenmem.db')) @@ -98,7 +105,7 @@ let _writesSinceRecallTraceSweep = 0 function getDb() { if (_db) return _db const Database = require('better-sqlite3') - _db = new Database(DB_PATH) + _db = new Database(resolveDbPath()) _db.pragma('journal_mode = WAL') _db.pragma('foreign_keys = ON') _db.pragma('busy_timeout = 5000') // wait 5s on concurrent writes instead of immediate error @@ -156,7 +163,7 @@ export function initMemory() { } } - log(`Initialized — DB at ${DB_PATH}`) + log(`Initialized — DB at ${resolveDbPath()}`) // ── FTS migration: if simple extension loaded but FTS uses old tokenizer, rebuild ── if (_simpleLoaded) { @@ -1046,6 +1053,10 @@ export async function recallForClients(o = {}) { deadlineMs: Number.isFinite(o.deadlineMs) && o.deadlineMs > 0 ? o.deadlineMs : null, _filterLevel: levels.length ? levels.join(',') : null, _minImportance: minImportance > 0 ? minImportance : null, + // The pool is over-fetched (up to 30) and trimmed below; bump access only + // for the rows actually returned, same contract as buildMemoryContext + // (#7). Without this every candidate got +1 per hook call. + _deferAccessBump: true, _out: out, }) @@ -1068,6 +1079,15 @@ export async function recallForClients(o = {}) { if (o.requireVec) memories = memories.filter(m => typeof m.vec_distance === 'number') memories = memories.slice(0, limit) + if (memories.length > 0) { + try { + const now = Date.now() + const stmt = getDb().prepare(`UPDATE memories SET last_accessed = ?, access_count = access_count + 1 WHERE rowid = ?`) + const tx = getDb().transaction((rows) => { for (const m of rows) stmt.run(now, m.rowid) }) + tx(memories) + } catch {} + } + // hit_count is the raw pool; final_hit_count is what survived filtering and // actually reached the caller. Without this the utilization stats read as if // every candidate got used. @@ -2467,7 +2487,14 @@ export async function recallMemoriesHybrid(opts = {}) { // importance keeps what it is good at: the min_importance filter, // surface-cold thresholds, display. Filtering on it is a caller stating a // floor; ranking on it is the store guessing. - const score = (rrf * 10 + freqScore * 0.10 + timeScore * 0.06) * decay + // + // freqScore is 0.02 for the same reason. It saturates at 20 accesses, so on a + // long-lived store it is effectively binary: a row surfaced 20+ times beat a + // fresh one by 0.10 — about 38 rank positions at RRF's spacing. Combined with + // recallForClients bumping its whole candidate pool, the loop fed itself: on + // one 10k-row store a single row reached 46% of two weeks of hook recalls. + // At 0.02 it stays a tiebreak (~8 positions at most). + const score = (rrf * 10 + freqScore * 0.02 + timeScore * 0.06) * decay const temporalMetadata = temporalWindow ? { temporal_match: isInTemporalWindow(row, temporalWindow) } : {} diff --git a/level-rank-offset.test.mjs b/level-rank-offset.test.mjs index 36c52cb..1018071 100644 --- a/level-rank-offset.test.mjs +++ b/level-rank-offset.test.mjs @@ -25,7 +25,7 @@ try { // Pins the composite's shape. importanceScore left the sum deliberately: a // self-rated field measured to be uncorrelated with use was buying 2-4 rank // positions against RRF's ~0.0026 spacing. See ranking-importance.integration.test.mjs. - const scoreMatch = hybridSource.match(/const score = \(rrf \* 10 \+ freqScore \* 0\.10 \+ timeScore \* 0\.06\) \* decay/) + const scoreMatch = hybridSource.match(/const score = \(rrf \* 10 \+ freqScore \* 0\.02 \+ timeScore \* 0\.06\) \* decay/) const levelWeightMatch = hybridSource.match(/const levelWeight = LEVEL_WEIGHT\[row\.memory_level\] \|\| 1\.0/) check('hybrid fusion declares the intended level rank offsets', diff --git a/package.json b/package.json index abb2fc2..f692e89 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "mneme", - "version": "2.11.0", + "version": "2.11.1", "description": "Token-efficient persistent memory for AI agents — SQLite + FTS5 + sqlite-vec hybrid search + MCP on-demand recall. Save 80-90% memory-related token costs.", "type": "module", "main": "index.mjs", From eb72de019ed5f1d0eb11d5195b2ec185b710aa3d Mon Sep 17 00:00:00 2001 From: MXAntian Date: Wed, 23 Sep 2026 11:31:27 +0800 Subject: [PATCH 2/2] fix(hook): one round trip via prefer_vec instead of require-then-retry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Self-review P1: the zero-row retry made every zero-config install (no embeddings) pay two sequential HTTP round trips on every trigger, and in the worst case the pair ate the whole safety timer before the CLI fallback could run. prefer_vec / --prefer-vec is the fail-open twin of require_vec: keep only vector-backed rows when any exist, otherwise return the unfiltered rows — one call either way. Servers older than 2.11.1 ignore the field and answer as before. Also: an unterminated (truncated upstream) is treated as non-user input instead of surviving whole as the query; the known trade-off of pasting text that starts with these tags is documented. Tests: access-bump-scope gains requireVec/preferVec cases on a no-embedding DB (0 rows vs FTS fallback); userPromptText gains the truncated case. Co-authored-by: 千夏 Co-Authored-By: Claude Opus 5.5 --- access-bump-scope.integration.test.mjs | 8 ++++++++ hooks/prompt-recall-trigger.mjs | 7 +++++-- hooks/prompt-recall-trigger.test.mjs | 2 ++ hooks/prompt-recall.mjs | 16 +++++++--------- index.mjs | 11 +++++++++++ mcp-server.mjs | 1 + 6 files changed, 34 insertions(+), 11 deletions(-) diff --git a/access-bump-scope.integration.test.mjs b/access-bump-scope.integration.test.mjs index b6ac1d3..f6b115d 100644 --- a/access-bump-scope.integration.test.mjs +++ b/access-bump-scope.integration.test.mjs @@ -50,6 +50,14 @@ check('every returned row was bumped', [...returned].every(id => bumped.includes check('no row outside the result was bumped', bumped.every(id => returned.has(id)), `bumped=${bumped} returned=${[...returned]}`) +// preferVec is the fail-open twin of requireVec. This temp DB has no embedding +// config, i.e. the zero-config install: requireVec must return nothing and +// preferVec must fall back to the FTS rows in the same call. +const strict = await recallForClients({ query: marker, limit: 2, minImportance: 1, requireVec: true, source: 'test' }) +const lenient = await recallForClients({ query: marker, limit: 2, minImportance: 1, preferVec: true, source: 'test' }) +check('requireVec without embeddings returns nothing', strict.hits.length === 0, `got ${strict.hits.length}`) +check('preferVec without embeddings falls back to FTS rows', lenient.hits.length === 2, `got ${lenient.hits.length}`) + closeMemory() console.log(`\n${fail === 0 ? 'PASS' : 'FAIL'}: ${pass} passed / ${fail} failed`) process.exit(fail === 0 ? 0 : 1) diff --git a/hooks/prompt-recall-trigger.mjs b/hooks/prompt-recall-trigger.mjs index 6cb85ed..a361d80 100644 --- a/hooks/prompt-recall-trigger.mjs +++ b/hooks/prompt-recall-trigger.mjs @@ -30,13 +30,16 @@ const TRIGGERS = [ // wrapper text, and those reports are dense with exactly the infrastructure // nouns the triggers below look for. // -// Returns '' when nothing user-authored remains. +// Returns '' when nothing user-authored remains. Known trade-off: a user who +// pastes text that itself starts with one of these tags gets no recall for it. const NOT_USER_INPUT = /^<(task-notification|cross-session-message)\b/ export function userPromptText(raw) { if (!raw || typeof raw !== 'string') return '' const text = raw.replace(/[\s\S]*?<\/system-reminder>/g, '').trim() - if (NOT_USER_INPUT.test(text)) return '' + // An unterminated block (truncated upstream) would otherwise survive whole + // and be sent as the query. + if (NOT_USER_INPUT.test(text) || text.startsWith(' { // Plain prompts pass through untouched. assert.equal(userPromptText(' how to restart the daemon '), 'how to restart the daemon') assert.equal(userPromptText(undefined), '') + // A truncated reminder is not user input either. + assert.equal(userPromptText('\nThe user started task X and the text was cut'), '') }) diff --git a/hooks/prompt-recall.mjs b/hooks/prompt-recall.mjs index 9e7b33f..b80f611 100644 --- a/hooks/prompt-recall.mjs +++ b/hooks/prompt-recall.mjs @@ -168,16 +168,18 @@ async function recallOverHttp(body) { } catch { return null } } async function runRecall(query, sessionId) { - const httpReq = (requireVec) => recallOverHttp({ + const viaHttp = await recallOverHttp({ query: query, limit: CFG.limit, min_importance: CFG.minImportance, level: CFG.level, - // Filter to rows with vector evidence BEFORE the server trims to `limit`. + // Prefer rows with vector evidence BEFORE the server trims to `limit`. // Otherwise the few slots go to character-level FTS and entity matches — // on CJK text that is mostly generic rows sharing one common noun — and the - // semantically relevant rows sit just below the cut. - require_vec: requireVec, + // semantically relevant rows sit just below the cut. Fail-open: with no + // embeddings the server returns plain FTS rows in the same round trip. + // (Servers older than 2.11.1 ignore the field and behave as before.) + prefer_vec: true, source: 'mneme-prompt-recall', session_id: sessionId, // Tell the server how long we will actually wait, so a slow embedding @@ -187,11 +189,6 @@ async function runRecall(query, sessionId) { // so the embedding gets httpTimeoutMs − 250 — 1250 ms at the default. deadline_ms: Math.max(300, CFG.httpTimeoutMs - 100), }) - // Zero rows under require_vec means no embeddings configured, the embedding - // call degraded, or genuinely nothing — ask again without it so zero-config - // installs keep the plain FTS behaviour (and the consensus gate below). - let viaHttp = await httpReq(true) - if (viaHttp && viaHttp.hits.length === 0) viaHttp = await httpReq(false) if (viaHttp) return viaHttp const args = [ @@ -202,6 +199,7 @@ async function runRecall(query, sessionId) { '--level', CFG.level, '--limit', String(CFG.limit), '--source', 'mneme-prompt-recall', + '--prefer-vec', '--session-id', sessionId, ] const r = spawnSync(process.execPath, args, { diff --git a/index.mjs b/index.mjs index 4b1b35d..d220440 100644 --- a/index.mjs +++ b/index.mjs @@ -1027,6 +1027,8 @@ function _parseEventTime(v) { * @param {number} [o.minImportance=0] importance floor (0 = no filter) * @param {string[]} [o.levels] memory_level allowlist * @param {boolean} [o.requireVec] keep only rows with vector evidence + * @param {boolean} [o.preferVec] keep only rows with vector evidence when any exist; + * otherwise return the unfiltered rows (no embeddings / degraded to FTS) * @param {string} [o.source] recall_log label * @param {string} [o.sessionId] recall_log session * @param {number} [o.deadlineMs] the caller's remaining budget in ms; bounds the @@ -1077,6 +1079,14 @@ export async function recallForClients(o = {}) { // callers need the vec rows to survive. With embedding down this yields 0 rows — // fail-closed is correct for inject paths. if (o.requireVec) memories = memories.filter(m => typeof m.vec_distance === 'number') + // preferVec: same filter, but fail-open. Lets a caller get vector-backed rows + // when embeddings are up and plain FTS rows when they are not, in one round + // trip — the alternative (require, then retry on zero) doubles latency on + // every zero-config install. + else if (o.preferVec) { + const withVec = memories.filter(m => typeof m.vec_distance === 'number') + if (withVec.length > 0) memories = withVec + } memories = memories.slice(0, limit) if (memories.length > 0) { @@ -4354,6 +4364,7 @@ if (_isMain) { const res = await recallForClients({ query, limit, minImportance, levels: levelFilter, requireVec: hasFlag('--require-vec'), + preferVec: hasFlag('--prefer-vec'), source: getFlag('--source') || 'cli', sessionId: getFlag('--session-id') || null, }) diff --git a/mcp-server.mjs b/mcp-server.mjs index a86d4df..e709035 100644 --- a/mcp-server.mjs +++ b/mcp-server.mjs @@ -663,6 +663,7 @@ if (useHttp) { minImportance: Number.isFinite(p.min_importance) ? p.min_importance : 0, levels: Array.isArray(p.levels) ? p.levels : (typeof p.level === 'string' && p.level ? p.level.split(',') : []), requireVec: !!p.require_vec, + preferVec: !!p.prefer_vec, // The caller's remaining budget. The server clamps its embedding // wait to fit, degrading to FTS *inside* the budget instead of the // caller aborting first and re-doing the work cold. See