diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b4a6389..e1494f2 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -66,6 +66,16 @@ jobs: TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-ranking-importance.db run: node ranking-importance.integration.test.mjs + - name: Integration test (access bump — only returned rows, not the candidate pool) + env: + TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-access-bump.db + run: node access-bump-scope.integration.test.mjs + + - name: Integration test (DB path read at open time, not import time) + env: + TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-late-db.db + run: node db-path-late-env.integration.test.mjs + - name: Integration test (encoding-damage detection, both write paths) env: TOKENMEM_DB_PATH: ${{ runner.temp }}/mneme-ci-encoding-damage.db diff --git a/access-bump-scope.integration.test.mjs b/access-bump-scope.integration.test.mjs new file mode 100644 index 0000000..f6b115d --- /dev/null +++ b/access-bump-scope.integration.test.mjs @@ -0,0 +1,63 @@ +// recallForClients over-fetches a candidate pool (up to 30 rows when filtering) +// and trims it to the caller's limit. Access counts must move only for the rows +// that were returned. When the whole pool was bumped, every hook call added +1 to +// ~30 rows while showing 3, and since access frequency feeds ranking and decay the +// rows that were already on top stayed there: on one 10k-row store a single row +// reached 46% of two weeks of hook recalls. +// +// Run: TOKENMEM_DB_PATH=/tmp/x.db node access-bump-scope.integration.test.mjs +import { initMemory, closeMemory, storeMemory, recallForClients } from './index.mjs' +import Database from 'better-sqlite3' + +const DB_PATH = process.env.TOKENMEM_DB_PATH +if (!DB_PATH) { console.error('FATAL: set TOKENMEM_DB_PATH'); process.exit(2) } + +let pass = 0, fail = 0 +const check = (label, cond, detail = '') => { + if (cond) { pass++; console.log(`✓ ${label}`) } + else { fail++; console.log(`✗ ${label}${detail ? ' — ' + detail : ''}`) } +} + +initMemory() +const marker = `zzbump${Math.floor(Math.random() * 1e6)}` +const ids = [] +for (let i = 0; i < 12; i++) { + ids.push(storeMemory({ + content: `${marker} pool row ${i} ${'x'.repeat(i)}`, + importance: 7, memoryLevel: 'semi_abstract', memoryType: 'long_term', + })) +} + +const readCounts = () => { + const db = new Database(DB_PATH, { readonly: true }) + const rows = db.prepare(`SELECT rowid, access_count FROM memories WHERE rowid IN (${ids.map(() => '?').join(',')})`).all(...ids) + db.close() + return new Map(rows.map(r => [String(r.rowid), r.access_count || 0])) +} + +const before = readCounts() +// min_importance > 0 is what makes recallForClients over-fetch — the shape the hooks send. +const res = await recallForClients({ query: marker, limit: 2, minImportance: 1, source: 'test' }) +const after = readCounts() + +const returned = new Set(res.hits.map(h => String(h.id))) +const bumped = [...after].filter(([id, n]) => n > (before.get(id) || 0)).map(([id]) => id) + +check('the pool held more candidates than were returned', res.hits.length === 2 && bumped.length <= 2, + `returned=${res.hits.length} bumped=${bumped.length}`) +check('every returned row was bumped', [...returned].every(id => bumped.includes(id)), + `returned=${[...returned]} bumped=${bumped}`) +check('no row outside the result was bumped', bumped.every(id => returned.has(id)), + `bumped=${bumped} returned=${[...returned]}`) + +// preferVec is the fail-open twin of requireVec. This temp DB has no embedding +// config, i.e. the zero-config install: requireVec must return nothing and +// preferVec must fall back to the FTS rows in the same call. +const strict = await recallForClients({ query: marker, limit: 2, minImportance: 1, requireVec: true, source: 'test' }) +const lenient = await recallForClients({ query: marker, limit: 2, minImportance: 1, preferVec: true, source: 'test' }) +check('requireVec without embeddings returns nothing', strict.hits.length === 0, `got ${strict.hits.length}`) +check('preferVec without embeddings falls back to FTS rows', lenient.hits.length === 2, `got ${lenient.hits.length}`) + +closeMemory() +console.log(`\n${fail === 0 ? 'PASS' : 'FAIL'}: ${pass} passed / ${fail} failed`) +process.exit(fail === 0 ? 0 : 1) diff --git a/db-path-late-env.integration.test.mjs b/db-path-late-env.integration.test.mjs new file mode 100644 index 0000000..f8f782e --- /dev/null +++ b/db-path-late-env.integration.test.mjs @@ -0,0 +1,47 @@ +// The DB path is read when the store opens, not when the module is imported. +// +// A host that imports mneme and then loads its own env file sets +// TOKENMEM_DB_PATH after ES import hoisting has already run this module's top +// level. With a module-level const the store silently opened the fallback DB +// beside index.mjs — and with more than one tenant on a machine, one tenant's +// writes landed in another's store. +// +// This file re-runs itself as a child with TOKENMEM_DB_PATH unset at import +// time, then sets it in the body — the host's order — and checks where the +// store actually opened. +// +// Run: TOKENMEM_DB_PATH=/tmp/x.db node db-path-late-env.integration.test.mjs +import { spawnSync } from 'node:child_process' +import { existsSync, unlinkSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +if (process.env.MNEME_LATE_DB_CHILD) { + const { initMemory, closeMemory, storeMemory } = await import('./index.mjs') + process.env.TOKENMEM_DB_PATH = process.env.MNEME_LATE_DB_CHILD // host sets it after import + initMemory() + storeMemory({ content: 'late env marker', importance: 5, memoryLevel: 'semi_abstract', memoryType: 'long_term' }) + closeMemory() + process.exit(0) +} + +const target = process.env.TOKENMEM_DB_PATH +if (!target) { console.error('FATAL: set TOKENMEM_DB_PATH'); process.exit(2) } +const latePath = target + '.late.db' +for (const sfx of ['', '-shm', '-wal']) { if (existsSync(latePath + sfx)) unlinkSync(latePath + sfx) } + +const env = { ...process.env, MNEME_LATE_DB_CHILD: latePath } +delete env.TOKENMEM_DB_PATH +delete env.MNEME_DB_PATH +const r = spawnSync(process.execPath, [fileURLToPath(import.meta.url)], { env, encoding: 'utf-8', timeout: 30000 }) + +let pass = 0, fail = 0 +const check = (label, cond, detail = '') => { + if (cond) { pass++; console.log(`✓ ${label}`) } + else { fail++; console.log(`✗ ${label}${detail ? ' — ' + detail : ''}`) } +} +check('child ran cleanly', r.status === 0, (r.stderr || '').slice(0, 300)) +check('store opened the path set after import', existsSync(latePath), latePath) + +for (const sfx of ['', '-shm', '-wal']) { try { unlinkSync(latePath + sfx) } catch {} } +console.log(`\n${fail === 0 ? 'PASS' : 'FAIL'}: ${pass} passed / ${fail} failed`) +process.exit(fail === 0 ? 0 : 1) diff --git a/docs/configuring-your-agent.md b/docs/configuring-your-agent.md index 18c5c80..ff6c2f8 100644 --- a/docs/configuring-your-agent.md +++ b/docs/configuring-your-agent.md @@ -234,9 +234,10 @@ be re-injected within one Claude Code session. | `MNEME_DB_PATH` | mneme's own `engram.db` | Alias for `TOKENMEM_DB_PATH`; picks the DB file | | `MNEME_INDEX_PATH` | `/index.mjs` | Override the engine entry point | | `MNEME_MIN_IMPORTANCE` | `6` | Floor for prompt-recall hits | -| `MNEME_LEVEL` | `meta_knowledge` | Prompt-recall level filter | +| `MNEME_LEVEL` | `meta_knowledge,semi_abstract` | Prompt-recall level filter | +| `MNEME_MAX_VEC_DISTANCE` | `0.95` | Prompt-recall drops hits farther than this when vectors are available | | `MNEME_LIMIT` | `5` | Prompt-recall candidate cap | -| `MNEME_MIN_CONSENSUS` | `2` | Prompt-recall skip if fewer hits | +| `MNEME_MIN_CONSENSUS` | `2` | Prompt-recall skip if fewer hits (only when no vector evidence) | | `MNEME_TOOL_MIN_IMPORTANCE` | `6` | Floor for tool-recall hits | | `MNEME_TOOL_LEVEL` | `meta_knowledge,semi_abstract` | Tool-recall level filter | | `MNEME_TOOL_LIMIT` | `4` | Tool-recall candidate cap | diff --git a/docs/configuring-your-agent.zh-CN.md b/docs/configuring-your-agent.zh-CN.md index b1b5eb8..d7d1ba0 100644 --- a/docs/configuring-your-agent.zh-CN.md +++ b/docs/configuring-your-agent.zh-CN.md @@ -206,9 +206,10 @@ MCP tool 是 pull 模式——Agent 自己决定何时 `recall_memory`。有些 | `MNEME_DB_PATH` | mneme 自带 `engram.db` | `TOKENMEM_DB_PATH` 的别名,选 DB 文件 | | `MNEME_INDEX_PATH` | `/index.mjs` | 覆盖引擎入口 | | `MNEME_MIN_IMPORTANCE` | `6` | prompt-recall 命中门槛 | -| `MNEME_LEVEL` | `meta_knowledge` | prompt-recall level 过滤 | +| `MNEME_LEVEL` | `meta_knowledge,semi_abstract` | prompt-recall level 过滤 | +| `MNEME_MAX_VEC_DISTANCE` | `0.95` | 有向量时,prompt-recall 丢弃距离大于此值的命中 | | `MNEME_LIMIT` | `5` | prompt-recall 候选上限 | -| `MNEME_MIN_CONSENSUS` | `2` | prompt-recall 少于此数就不注入 | +| `MNEME_MIN_CONSENSUS` | `2` | prompt-recall 少于此数就不注入(仅在没有向量证据时生效) | | `MNEME_TOOL_MIN_IMPORTANCE` | `6` | tool-recall 命中门槛 | | `MNEME_TOOL_LEVEL` | `meta_knowledge,semi_abstract` | tool-recall level 过滤 | | `MNEME_TOOL_LIMIT` | `4` | tool-recall 候选上限 | diff --git a/hooks/prompt-recall-trigger.mjs b/hooks/prompt-recall-trigger.mjs index 33e456a..a361d80 100644 --- a/hooks/prompt-recall-trigger.mjs +++ b/hooks/prompt-recall-trigger.mjs @@ -22,6 +22,27 @@ const TRIGGERS = [ /(?:路径|目录|位置|端口|凭证|密钥|环境变量|配置|脚本|工具|命令|启动|重启|守护进程)/, ] +// UserPromptSubmit carries more than what the user typed. In Claude Code, +// background-agent reports arrive as , messages from other +// sessions as , and app notices get prepended as +// . Recalling on the raw prompt means recalling on that +// wrapper text: on one store, 631 of 688 fast-path queries over two weeks were +// wrapper text, and those reports are dense with exactly the infrastructure +// nouns the triggers below look for. +// +// Returns '' when nothing user-authored remains. Known trade-off: a user who +// pastes text that itself starts with one of these tags gets no recall for it. +const NOT_USER_INPUT = /^<(task-notification|cross-session-message)\b/ + +export function userPromptText(raw) { + if (!raw || typeof raw !== 'string') return '' + const text = raw.replace(/[\s\S]*?<\/system-reminder>/g, '').trim() + // An unterminated block (truncated upstream) would otherwise survive whole + // and be sent as the query. + if (NOT_USER_INPUT.test(text) || text.startsWith(' 1500) return false diff --git a/hooks/prompt-recall-trigger.test.mjs b/hooks/prompt-recall-trigger.test.mjs index 7252d53..be44000 100644 --- a/hooks/prompt-recall-trigger.test.mjs +++ b/hooks/prompt-recall-trigger.test.mjs @@ -16,3 +16,22 @@ test('short steering and long pasted text stay outside the gate', () => { assert.equal(shouldTriggerPromptRecall('推进'), false) assert.equal(shouldTriggerPromptRecall('召回'.repeat(751)), false) }) + +test('only user-authored text reaches the trigger and the query', async () => { + const { userPromptText } = await import('./prompt-recall-trigger.mjs') + // App notice prepended to a real prompt: the notice goes, the prompt stays. + assert.equal( + userPromptText('\nThe user started background task X ("fix daemon restart")\n\nwhy is recall noisy'), + 'why is recall noisy', + ) + // Background-agent reports and other sessions' messages are not user input, + // even though they are full of the nouns the triggers look for. + assert.equal(userPromptText('\ndaemon config path port restart\n'), '') + assert.equal(userPromptText('where is the config'), '') + assert.equal(shouldTriggerPromptRecall(userPromptText('how to restart the daemon')), false) + // Plain prompts pass through untouched. + assert.equal(userPromptText(' how to restart the daemon '), 'how to restart the daemon') + assert.equal(userPromptText(undefined), '') + // A truncated reminder is not user input either. + assert.equal(userPromptText('\nThe user started task X and the text was cut'), '') +}) diff --git a/hooks/prompt-recall.mjs b/hooks/prompt-recall.mjs index 64bd80f..b80f611 100644 --- a/hooks/prompt-recall.mjs +++ b/hooks/prompt-recall.mjs @@ -25,7 +25,9 @@ // MNEME_DB_PATH alias for TOKENMEM_DB_PATH; where mneme's engram.db lives // MNEME_INDEX_PATH override for index.mjs (default: ../index.mjs) // MNEME_MIN_IMPORTANCE floor for hits (default: 6) -// MNEME_LEVEL recall level filter (default: meta_knowledge) +// MNEME_LEVEL recall level filter (default: meta_knowledge,semi_abstract) +// MNEME_MAX_VEC_DISTANCE drop hits farther than this when the server returned +// vector evidence (default: 0.95; ignored without embeddings) // MNEME_LIMIT max recall candidates (default: 5) // MNEME_MIN_CONSENSUS hide injection if hits < this (default: 2) // MNEME_STATE_DIR session-dedup file dir (default: ~/.claude/hooks) @@ -43,7 +45,7 @@ import { spawnSync } from 'node:child_process' import { readFileSync, writeFileSync, existsSync, mkdirSync, renameSync } from 'node:fs' import { resolve, dirname } from 'node:path' import { fileURLToPath } from 'node:url' -import { shouldTriggerPromptRecall } from './prompt-recall-trigger.mjs' +import { shouldTriggerPromptRecall, userPromptText } from './prompt-recall-trigger.mjs' const __dirname = dirname(fileURLToPath(import.meta.url)) const HOME = process.env.USERPROFILE || process.env.HOME || __dirname @@ -98,7 +100,12 @@ const CFG = { httpTimeoutMs: intEnv('MNEME_HTTP_TIMEOUT_MS', 1500), indexPath: process.env.MNEME_INDEX_PATH || resolve(__dirname, '..', 'index.mjs'), minImportance: intEnv('MNEME_MIN_IMPORTANCE', 6), - level: process.env.MNEME_LEVEL || 'meta_knowledge', + // semi_abstract is in the default on purpose. The triggers are operational + // (paths, ports, restarts, config), and the write-time meta gate downgrades + // anything carrying a path, port or version to semi_abstract — so under a + // meta-only filter the answers this hook exists to find could never match. + level: process.env.MNEME_LEVEL || 'meta_knowledge,semi_abstract', + maxVecDistance: (() => { const n = Number(process.env.MNEME_MAX_VEC_DISTANCE); return Number.isFinite(n) && n > 0 ? n : 0.95 })(), limit: intEnv('MNEME_LIMIT', 5), minConsensus: intEnv('MNEME_MIN_CONSENSUS', 2), stateDir: process.env.MNEME_STATE_DIR || resolve(HOME, '.claude', 'hooks'), @@ -166,6 +173,13 @@ async function runRecall(query, sessionId) { limit: CFG.limit, min_importance: CFG.minImportance, level: CFG.level, + // Prefer rows with vector evidence BEFORE the server trims to `limit`. + // Otherwise the few slots go to character-level FTS and entity matches — + // on CJK text that is mostly generic rows sharing one common noun — and the + // semantically relevant rows sit just below the cut. Fail-open: with no + // embeddings the server returns plain FTS rows in the same round trip. + // (Servers older than 2.11.1 ignore the field and behave as before.) + prefer_vec: true, source: 'mneme-prompt-recall', session_id: sessionId, // Tell the server how long we will actually wait, so a slow embedding @@ -185,6 +199,7 @@ async function runRecall(query, sessionId) { '--level', CFG.level, '--limit', String(CFG.limit), '--source', 'mneme-prompt-recall', + '--prefer-vec', '--session-id', sessionId, ] const r = spawnSync(process.execPath, args, { @@ -219,7 +234,7 @@ process.stdin.on('end', async () => { try { payload = JSON.parse(input || '{}') } catch { process.exit(0) } const sessionId = payload.session_id || payload.sessionId || 'unknown' - const prompt = (payload.prompt || '').trim() + const prompt = userPromptText(payload.prompt || '') if (!shouldTriggerPromptRecall(prompt)) process.exit(0) @@ -227,8 +242,16 @@ process.stdin.on('end', async () => { const recalled = await runRecall(query, sessionId) if (!recalled || !Array.isArray(recalled.hits)) process.exit(0) - const hits = recalled.hits - if (hits.length < CFG.minConsensus) process.exit(0) + // With vector evidence present, relevance is measurable — gate on it and + // skip the count heuristic. Without it (no embeddings), fall back to + // requiring agreement between several FTS hits. + let hits = recalled.hits + if (hits.some(h => typeof h.vec_distance === 'number')) { + hits = hits.filter(h => typeof h.vec_distance === 'number' && h.vec_distance <= CFG.maxVecDistance) + if (hits.length === 0) process.exit(0) + } else if (hits.length < CFG.minConsensus) { + process.exit(0) + } const injected = loadInjected(sessionId) const fresh = hits.filter(h => !injected.has(h.id)) diff --git a/index.mjs b/index.mjs index b0937bc..d220440 100644 --- a/index.mjs +++ b/index.mjs @@ -64,7 +64,14 @@ try { // the pre-rename era — silent migration would create a split-brain second // DB; we'd rather keep using the populated one) // 3. tokenmem.db (default for fresh installs) -const DB_PATH = process.env.TOKENMEM_DB_PATH +// +// Resolved at call time, not import time. A host that imports this module and +// then loads its own env file sets TOKENMEM_DB_PATH too late for a module-level +// const — ES imports are hoisted above the host's body — so the store silently +// opened the fallback DB. With more than one tenant per machine that means +// writes land in the wrong store. getDb() first runs inside initMemory(), after +// the host's env is ready. +const resolveDbPath = () => process.env.TOKENMEM_DB_PATH || (existsSync(resolve(__dirname, 'engram.db')) ? resolve(__dirname, 'engram.db') : resolve(__dirname, 'tokenmem.db')) @@ -98,7 +105,7 @@ let _writesSinceRecallTraceSweep = 0 function getDb() { if (_db) return _db const Database = require('better-sqlite3') - _db = new Database(DB_PATH) + _db = new Database(resolveDbPath()) _db.pragma('journal_mode = WAL') _db.pragma('foreign_keys = ON') _db.pragma('busy_timeout = 5000') // wait 5s on concurrent writes instead of immediate error @@ -156,7 +163,7 @@ export function initMemory() { } } - log(`Initialized — DB at ${DB_PATH}`) + log(`Initialized — DB at ${resolveDbPath()}`) // ── FTS migration: if simple extension loaded but FTS uses old tokenizer, rebuild ── if (_simpleLoaded) { @@ -1020,6 +1027,8 @@ function _parseEventTime(v) { * @param {number} [o.minImportance=0] importance floor (0 = no filter) * @param {string[]} [o.levels] memory_level allowlist * @param {boolean} [o.requireVec] keep only rows with vector evidence + * @param {boolean} [o.preferVec] keep only rows with vector evidence when any exist; + * otherwise return the unfiltered rows (no embeddings / degraded to FTS) * @param {string} [o.source] recall_log label * @param {string} [o.sessionId] recall_log session * @param {number} [o.deadlineMs] the caller's remaining budget in ms; bounds the @@ -1046,6 +1055,10 @@ export async function recallForClients(o = {}) { deadlineMs: Number.isFinite(o.deadlineMs) && o.deadlineMs > 0 ? o.deadlineMs : null, _filterLevel: levels.length ? levels.join(',') : null, _minImportance: minImportance > 0 ? minImportance : null, + // The pool is over-fetched (up to 30) and trimmed below; bump access only + // for the rows actually returned, same contract as buildMemoryContext + // (#7). Without this every candidate got +1 per hook call. + _deferAccessBump: true, _out: out, }) @@ -1066,8 +1079,25 @@ export async function recallForClients(o = {}) { // callers need the vec rows to survive. With embedding down this yields 0 rows — // fail-closed is correct for inject paths. if (o.requireVec) memories = memories.filter(m => typeof m.vec_distance === 'number') + // preferVec: same filter, but fail-open. Lets a caller get vector-backed rows + // when embeddings are up and plain FTS rows when they are not, in one round + // trip — the alternative (require, then retry on zero) doubles latency on + // every zero-config install. + else if (o.preferVec) { + const withVec = memories.filter(m => typeof m.vec_distance === 'number') + if (withVec.length > 0) memories = withVec + } memories = memories.slice(0, limit) + if (memories.length > 0) { + try { + const now = Date.now() + const stmt = getDb().prepare(`UPDATE memories SET last_accessed = ?, access_count = access_count + 1 WHERE rowid = ?`) + const tx = getDb().transaction((rows) => { for (const m of rows) stmt.run(now, m.rowid) }) + tx(memories) + } catch {} + } + // hit_count is the raw pool; final_hit_count is what survived filtering and // actually reached the caller. Without this the utilization stats read as if // every candidate got used. @@ -2467,7 +2497,14 @@ export async function recallMemoriesHybrid(opts = {}) { // importance keeps what it is good at: the min_importance filter, // surface-cold thresholds, display. Filtering on it is a caller stating a // floor; ranking on it is the store guessing. - const score = (rrf * 10 + freqScore * 0.10 + timeScore * 0.06) * decay + // + // freqScore is 0.02 for the same reason. It saturates at 20 accesses, so on a + // long-lived store it is effectively binary: a row surfaced 20+ times beat a + // fresh one by 0.10 — about 38 rank positions at RRF's spacing. Combined with + // recallForClients bumping its whole candidate pool, the loop fed itself: on + // one 10k-row store a single row reached 46% of two weeks of hook recalls. + // At 0.02 it stays a tiebreak (~8 positions at most). + const score = (rrf * 10 + freqScore * 0.02 + timeScore * 0.06) * decay const temporalMetadata = temporalWindow ? { temporal_match: isInTemporalWindow(row, temporalWindow) } : {} @@ -4327,6 +4364,7 @@ if (_isMain) { const res = await recallForClients({ query, limit, minImportance, levels: levelFilter, requireVec: hasFlag('--require-vec'), + preferVec: hasFlag('--prefer-vec'), source: getFlag('--source') || 'cli', sessionId: getFlag('--session-id') || null, }) diff --git a/level-rank-offset.test.mjs b/level-rank-offset.test.mjs index 36c52cb..1018071 100644 --- a/level-rank-offset.test.mjs +++ b/level-rank-offset.test.mjs @@ -25,7 +25,7 @@ try { // Pins the composite's shape. importanceScore left the sum deliberately: a // self-rated field measured to be uncorrelated with use was buying 2-4 rank // positions against RRF's ~0.0026 spacing. See ranking-importance.integration.test.mjs. - const scoreMatch = hybridSource.match(/const score = \(rrf \* 10 \+ freqScore \* 0\.10 \+ timeScore \* 0\.06\) \* decay/) + const scoreMatch = hybridSource.match(/const score = \(rrf \* 10 \+ freqScore \* 0\.02 \+ timeScore \* 0\.06\) \* decay/) const levelWeightMatch = hybridSource.match(/const levelWeight = LEVEL_WEIGHT\[row\.memory_level\] \|\| 1\.0/) check('hybrid fusion declares the intended level rank offsets', diff --git a/mcp-server.mjs b/mcp-server.mjs index a86d4df..e709035 100644 --- a/mcp-server.mjs +++ b/mcp-server.mjs @@ -663,6 +663,7 @@ if (useHttp) { minImportance: Number.isFinite(p.min_importance) ? p.min_importance : 0, levels: Array.isArray(p.levels) ? p.levels : (typeof p.level === 'string' && p.level ? p.level.split(',') : []), requireVec: !!p.require_vec, + preferVec: !!p.prefer_vec, // The caller's remaining budget. The server clamps its embedding // wait to fit, degrading to FTS *inside* the budget instead of the // caller aborting first and re-doing the work cold. See diff --git a/package.json b/package.json index abb2fc2..f692e89 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "mneme", - "version": "2.11.0", + "version": "2.11.1", "description": "Token-efficient persistent memory for AI agents — SQLite + FTS5 + sqlite-vec hybrid search + MCP on-demand recall. Save 80-90% memory-related token costs.", "type": "module", "main": "index.mjs",