From dfacf75bb452980413f519f09d8b9dd0f3c6cafa Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Mon, 28 Sep 2026 20:35:26 +0300 Subject: [PATCH] feat(search): O1 rare-identifier x100 boost at pre-rerank pool --- KNOWN_ISSUES.md | 5 + scripts/o1_holdout_gate.py | 144 +++++++++++++++++++++++++++++ src/core/search/engine.py | 148 ++++++++++++++++++++++++++++++ tests/test_o1_identifier_boost.py | 146 +++++++++++++++++++++++++++++ 4 files changed, 443 insertions(+) create mode 100644 scripts/o1_holdout_gate.py create mode 100644 tests/test_o1_identifier_boost.py diff --git a/KNOWN_ISSUES.md b/KNOWN_ISSUES.md index 082e4950..da484075 100644 --- a/KNOWN_ISSUES.md +++ b/KNOWN_ISSUES.md @@ -5,6 +5,11 @@ --- +## 2026-09-28 — Ретриевер-замеры без сброса реранкер-кэша недействительны (Open) + +- **Правило:** все retriever-замеры и A/B-тесты — только в свежем процессе либо с явным сбросом реранкер-кэша (`Searcher._reranker_cache.clear()`). Ключ кэша включает текст запроса (engine.py:1646): повтор того же запроса в том же процессе отдаёт закэшированные скоры, а не измеряет код. +- **Эвристика void-замера:** wall <2s на `hybrid_search_async` при ожидании полного пайплайна (embed+BM25+FTS+rerank) = подозрение на cache hit; сверяться с `Searcher._last_rerank_timing` (пусто = реранкер не работал). Холодный FTS-билд (~2.5s) — обратная ловушка: ПЕРВЫЙ замер в свежем процессе молча теряет FTS-тир (2s `wait_for`), нужен discarded warm-up на чужом запросе. +- **Статус:** 🟡 Open (процедурное правило; guard-скрипт `scripts/o1_holdout_gate.py` — fresh-process + warm-up + void-флаг). **21 entries** — compressed per §4.8 R3 (conclusion-first; dedup 2026-09-08, 2026-09-21). Closed entries moved to docs/archive/KNOWN_ISSUES_2026_09.md on 2026-09-27 (R1 size guard; second batch on merge experiment/4a-unit-of-return). diff --git a/scripts/o1_holdout_gate.py b/scripts/o1_holdout_gate.py new file mode 100644 index 00000000..5b6a7125 --- /dev/null +++ b/scripts/o1_holdout_gate.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""O1 holdout gate (live, fresh-process): P2 + H1-H12 + N/doc controls. + +Usage: + python scripts/o1_holdout_gate.py [--project D:/Project/MSCodeBase] + +Each query runs hybrid_search_async(limit=5) in THIS fresh process +(reranker cache starts empty -> no cache-hit void measurements). +Prints a result table; exit 1 on gate failure (see PASS RULES below). + +Holdout set note: no canonical H1-H12 exists on main (verified 2026-09-28: +grep finds only unrelated H-labels in diary/archive). H1-H12 below are +O1-holdout-v1, defined HERE (multi-token identifier queries with known +target files, disjoint from the frozen eval-16 calibration use). +""" +import argparse +import asyncio +import sys +import time +import traceback +from pathlib import Path + +if sys.stdout.encoding and sys.stdout.encoding.lower() != "utf-8": + try: + sys.stdout.reconfigure(encoding="utf-8") + except Exception: # noqa: BLE001 — encoding guard must never fail startup + pass + +ROOT = Path(__file__).resolve().parent.parent +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +P2 = {"id": "P2", "query": "hybrid_search_async reciprocal_rank_fusion FTS5 BM25", + "target": "src/core/search/engine.py", "expect_rank": 1} + +HOLDOUT = [ + {"id": "H1", "query": "hybrid_search_async BM25 vector FTS5 ranking", + "target": "src/core/search/engine.py"}, + {"id": "H2", "query": "BM25 common term index search query ranking", + "target": None, "expect_no_boost": True}, # collision control + {"id": "H3", "query": "ArtifactGC 30d projects 90d telemetry retention", + "target": "src/core/artifact_gc.py"}, + {"id": "H4", "query": "resolve_ubatch embed 512 rerank 1024 batch", + "target": "src/providers/reranker/llama_install.py"}, + {"id": "H5", "query": "_is_pid_alive OpenProcess SYNCHRONIZE stale PID guard", + "target": "src/providers/reranker/llama_runner.py"}, + {"id": "H6", "query": "add_node add_edge SQLite transaction per node PropertyGraph", + "target": "src/core/graph.py"}, + {"id": "H7", "query": "LLAMA_EMBED_MAX_TOKENS truncates 480 e5-small context", + "target": "src/providers/embedder/remote_embedder.py"}, + {"id": "H8", "query": "json_group_array collect Cypher cypher_sql translation", + "target": "src/core/search/cypher_sql.py"}, + {"id": "H9", "query": "bootstrap_tests TESTS edges verifying tests far better", + "target": "src/core/bootstrap_tests.py"}, + {"id": "H10", "query": "to_pandas columns lancedb FreshnessChecker broken dead code", + "target": "src/core/indexing/freshness.py"}, + {"id": "H11", "query": "CodeParser tree_sitter parse_file AST chunks walk", + "target": "src/core/indexing/parser.py"}, + {"id": "H12", "query": "_cleanup_old_projects 30d retention_policy ArtifactGC", + "target": "src/core/artifact_gc.py"}, + {"id": "N1", "query": "quantum computing configuration", + "target": None, "expect_no_boost": True}, # out-of-domain negative + {"id": "DOC", "query": "begin_write _write_lock RLock reindex freeze server", + "target": "src/core/indexing/db_writer.py", "expect_no_doc_boost": True}, +] + + +def build_searcher(project: Path): + from src.core.di_container import IndexerFactoryKey, create_service_collection + + services = create_service_collection(project) + factory = services.resolve(IndexerFactoryKey) + indexer = factory(project) + return indexer.searcher + + +def rank_of(results, target): + for i, r in enumerate(results, 1): + if (r.get("metadata") or {}).get("file") == target: + return i + return None + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--project", default=str(ROOT)) + args = ap.parse_args() + project = Path(args.project) + try: + import subprocess # noqa: PLC0415 + rev = subprocess.run(["git", "rev-parse", "--short", "HEAD"], + cwd=str(ROOT), capture_output=True, + timeout=10).stdout.decode().strip() + except Exception: # noqa: BLE001 — git metadata is best-effort diagnostics + rev = "unknown" + print(f"O1 holdout gate | rev={rev} | project={project} | fresh process") + searcher = build_searcher(project) + # Warm-up (discarded): cold FTS5 to_pandas build exceeds the 2s tier budget + # on main and would silently drop the FTS tier for the FIRST measured query + # (known issue, PR52 territory). Warm-up uses a disjoint query -> no + # reranker-cache contamination (cache key includes the query text). + asyncio.run(searcher.hybrid_search_async("warmup cold start primer", limit=3)) + rows = [] + for case in [P2, *HOLDOUT]: + t0 = time.perf_counter() + results = asyncio.run(searcher.hybrid_search_async(case["query"], limit=5)) + wall = time.perf_counter() - t0 + boosted = [r for r in results if r.get("identifier_boost")] + doc_boosted = [r for r in boosted + if str((r.get("metadata") or {}).get("file", "")) + .lower().endswith((".md", ".markdown", ".rst", ".txt", ".ipynb"))] + rank = rank_of(results, case["target"]) if case.get("target") else None + void = wall < 2.0 and bool(getattr(searcher, "_last_rerank_timing", None) == {}) + rows.append({"id": case["id"], "rank": rank, "target": case.get("target"), + "n_boost": len(boosted), "n_doc_boost": len(doc_boosted), + "wall": round(wall, 2), "void": void}) + print(f"{case['id']:>3} rank={rank} target={case.get('target')} " + f"boost={len(boosted)} doc_boost={len(doc_boosted)} wall={wall:.2f}s") + print("---") + fails = [] + p2 = rows[0] + if p2["rank"] != 1: + fails.append(f"P2 rank={p2['rank']} (expected 1)") + for row in rows: + case = next(c for c in [P2, *HOLDOUT] if c["id"] == row["id"]) + if case.get("expect_no_boost") and row["n_boost"]: + fails.append(f"{row['id']} unexpectedly boosted") + if case.get("expect_no_doc_boost") and row["n_doc_boost"]: + fails.append(f"{row['id']} doc chunk boosted") + if row["void"]: + fails.append(f"{row['id']} reranker-cache void (wall<2s, no timing)") + if fails: + print("GATE FAIL: " + "; ".join(fails)) + return 1 + print("GATE PASS") + return 0 + + +if __name__ == "__main__": + try: + sys.exit(main()) + except Exception: # noqa: BLE001 — gate must report, never crash silently + traceback.print_exc() + sys.exit(1) diff --git a/src/core/search/engine.py b/src/core/search/engine.py index e8d012b6..1cf010cb 100644 --- a/src/core/search/engine.py +++ b/src/core/search/engine.py @@ -43,6 +43,7 @@ _extract_key_terms, _extract_symbol_name, _filter_by_time, + _tokenize, ) _sync_executor = concurrent.futures.ThreadPoolExecutor( @@ -135,6 +136,112 @@ def _boost_exact_name_matches(results: List[dict], query: str) -> List[dict]: return results +# O1 (2026-09-28): identifier boost for MULTI-token queries at pre-rerank pool. +# Single-identifier queries are served by _boost_exact_name_matches (+ graph +# short-circuit); multi-token queries ("hybrid_search_async reciprocal_rank_fusion +# FTS5 BM25") never reach it (_IDENTIFIER_QUERY_RE rejects spaces), so a rare +# identifier inside them drowned in multi-tier RRF noise. Red-team defenses +# (mandatory acceptance criteria): +# D1 rarity gate — candidate only if it is the STRICTLY rarest query term +# (df from existing BM25 stats, no new index structures); a common-term +# collision (H2-style: `BM25` frequent in index) never boosts. +# D2 exact gate — full-token equality token == symbol_name, never substring; +# _is_doc_chunk exclusion inherited (docs never boosted). +# D3 pre-rerank application — runs on the pre-rerank pool (before +# _apply_multi_reranker_async), never post-cut; the x100 only guarantees +# pool survival past the reranker's top_n cut (reranker re-sorts itself). +# Shape classes mirror CodeJury _identifiers: --flag, UPPER, _ in name, CamelCase. +_IDENTIFIER_TOKEN_RES = ( + re.compile(r"^--[A-Za-z][A-Za-z0-9_-]*$"), # --flag + re.compile(r"^[A-Z][A-Z0-9_]{1,63}$"), # CONSTANT / BM25 / FTS5 + re.compile(r"^[A-Za-z_][A-Za-z0-9_]*_[A-Za-z0-9_]+$"), # snake_case + re.compile(r"^[a-z]+[A-Z][A-Za-z0-9]*$"), # camelCase + re.compile(r"^[A-Z][a-z0-9]+[A-Z][A-Za-z0-9]*$"), # PascalCase +) + +# Proven x100 factor convention (cf. _boost_exact_name_matches). +_O1_BOOST_FACTOR = 100.0 + +# Same split pattern as Searcher._tokenizer_re: BM25 terms and O1 query terms +# must share tokenization, otherwise df comparison is meaningless. +_O1_TOKENIZER_RE = re.compile(r"\W+") + + +def _extract_identifier_tokens(query: str) -> List[str]: + """Identifier-shaped raw tokens of the query (O1/D1 shape classes).""" + toks = re.findall( + r"--[A-Za-z][A-Za-z0-9_-]*|[A-Za-z_][A-Za-z0-9_]*", query or "" + ) + return [t for t in toks if any(rx.match(t) for rx in _IDENTIFIER_TOKEN_RES)] + + +def _identifier_norm(tok: str) -> str: + """Normalized form comparable with BM25 terms (lowercase, no leading dashes).""" + return tok.lower().lstrip("-") + + +def _pick_rare_identifier(query: str, df_of) -> Optional[str]: + """O1 candidate: identifier-shaped token that is the STRICTLY rarest query term. + + Args: + query: raw multi-token query. + df_of: callable term -> int, document frequency from existing BM25 stats. + + Returns: + Normalized candidate or None. None covers: no identifier-shaped token, + single-term query (served by _boost_exact_name_matches), tie for rarest, + and H2-style collision (identifier-shaped but common term). + """ + idents = _extract_identifier_tokens(query) + if not idents: + return None + qtokens = _tokenize(query, _O1_TOKENIZER_RE) + uniq = set(qtokens) + if len(uniq) < 2: + return None + df_cache: Dict[str, int] = {} + + def df(t: str) -> int: + if t not in df_cache: + try: + df_cache[t] = int(df_of(t) or 0) + except Exception: # noqa: BLE001 — broken stats never break search + df_cache[t] = 0 + return df_cache[t] + + for raw in idents: + norm = _identifier_norm(raw) + others = uniq - {norm} + if not others: + continue + if df(norm) < min(df(t) for t in others): + return norm + return None + + +def _identifier_exact_match(r: Dict, candidate_norm: str) -> bool: + """True only on full-token equality symbol == candidate (O1/D2, never substring).""" + if _is_doc_chunk(r.get("metadata") or {}): + return False + meta = r.get("metadata") or {} + text = str(r.get("text", "") or "") + sym = str(meta.get("symbol_name", "") or "") + if not sym: + sym = _extract_symbol_name(text) + return bool(sym) and sym.lower() == candidate_norm + + +def _boost_rare_identifier(pool: List[dict], candidate_norm: str) -> List[dict]: + """Applies the O1 x100 boost to exact symbol matches inside the pool.""" + for r in pool: + if _identifier_exact_match(r, candidate_norm): + r["final_score"] = (r.get("final_score", 0) or 0) * _O1_BOOST_FACTOR + r["identifier_boost"] = True + # Stable: boosted first (input order), rest follow (input order). + pool.sort(key=lambda r: 0 if r.get("identifier_boost") else 1) + return pool + + def _prepend_code_name_matches( results: List[dict], pool: Optional[List[dict]], query: str, limit: int ) -> List[dict]: @@ -517,6 +624,38 @@ def hybrid_search( ) ) + def _apply_o1_identifier_boost( + self, pool: List[dict], query: str + ) -> List[dict]: + """O1: rare-identifier x100 boost on the pre-rerank pool (defenses D1-D3). + + Builds df from the EXISTING BM25 stats (no new index structures), + picks the strictly-rarest identifier-shaped query token and boosts + exact symbol matches. Any failure degrades to unboosted pool. + """ + if not pool: + return pool + try: + self._build_bm25_index() + except Exception: # noqa: BLE001 — degraded BM25 never breaks search + return pool + bm25 = getattr(self, "_bm25", None) or {} + if not bm25: + return pool + + def df_of(term: str) -> int: + n = 0 + for doc_terms in bm25.values(): + if term in doc_terms: + n += 1 + return n + + candidate = _pick_rare_identifier(query, df_of) + if not candidate: + return pool + logger.debug(f"[O1] rare-identifier boost: '{candidate}' x{_O1_BOOST_FACTOR:g}") + return _boost_rare_identifier(pool, candidate) + async def hybrid_search_async( self, query: str, @@ -757,6 +896,15 @@ async def hybrid_search_async( if tracer and _mmr_before: tracer.record_mmr(_mmr_before, pre_rerank_results, lambda_param=0.6) + # === O1 (2026-09-28): rare-identifier boost at PRE-RERANK pool (D3) === + # Runs here — after sort+cut/MMR, before the reranker — never post-cut: + # the x100 guarantees a rare exact symbol match survives the reranker's + # top_n cut; the reranker itself still re-sorts by its own scores. + # (On PR52 this site hosts anchor_tier_winners; O1 composes after it.) + pre_rerank_results = self._apply_o1_identifier_boost( + pre_rerank_results, query + ) + # Мульти-провайдерный реранкинг (Ollama / LM Studio) — опциональный # Реранкер перезаписывает final_score своими семантическими весами _pre_rerank = list(pre_rerank_results) if tracer else None diff --git a/tests/test_o1_identifier_boost.py b/tests/test_o1_identifier_boost.py new file mode 100644 index 00000000..fe327192 --- /dev/null +++ b/tests/test_o1_identifier_boost.py @@ -0,0 +1,146 @@ +"""O1 rare-identifier boost: unit tests (spec items 1-3, 5). + +Covers: +- extraction of identifier-shaped tokens (snake/CamelCase/Pascal/CONSTANT/--flag); +- D1 rarity gate incl. H2-style collision (common identifier-shaped term -> None); +- D2 exact gate (full-token equality only, never substring) + doc-guard control; +- D3 pre-rerank pool assertion (gold in pool pre-boost, first post-boost, x100). +""" +from unittest.mock import MagicMock + +from src.core.search.engine import ( + Searcher, + _boost_rare_identifier, + _extract_identifier_tokens, + _identifier_exact_match, + _pick_rare_identifier, +) + + +def _chunk(symbol=None, text=None, score=1.0, fname="src/core/search/engine.py"): + return { + "text": text if text is not None else f"async def {symbol}(): ...", + "metadata": {"file": fname, "chunk_index": 0, **( + {"symbol_name": symbol} if symbol else {})}, + "final_score": score, + } + + +def _doc_chunk(symbol): + return { + "text": f"class {symbol} described here", + "metadata": {"file": "KNOWN_ISSUES.md", "chunk_index": 0, + "symbol_name": symbol}, + "final_score": 5.0, + } + + +# --- extraction (spec 1) --- +def test_extract_shapes(): + toks = _extract_identifier_tokens( + "how does hybrid_search_async handle CamelCase PascalCase FTS5 --max-tokens query") + assert "hybrid_search_async" in toks + assert "CamelCase" in toks + assert "PascalCase" in toks + assert "FTS5" in toks + assert "--max-tokens" in toks + + +def test_extract_plain_words_ignored(): + assert _extract_identifier_tokens("how does search handle query") == [] + + +# --- D1 rarity gate (spec 1) --- +def test_pick_rarest_identifier(): + df = {"hybrid_search_async": 2, "search": 50, "query": 40, "how": 90} + assert _pick_rare_identifier( + "how hybrid_search_async search query", df.get) == "hybrid_search_async" + + +def test_collision_common_term_no_boost_h2(): + """H2-style: identifier-shaped `BM25` is common in index -> must NOT boost.""" + df = {"bm25": 80, "search": 50, "hybrid_search_async": 2, "query": 40} + # BM25 itself is not the rarest -> None for a BM25-anchored query... + assert _pick_rare_identifier("BM25 search query", df.get) is None + # ...and a query where the only identifier is common also yields None. + assert _pick_rare_identifier( + "BM25 search query hybrid", {**df, "hybrid": 90}.get) is None + + +def test_tie_for_rarest_no_boost(): + df = {"hybrid_search_async": 5, "reciprocal_rank_fusion": 5, "query": 40} + assert _pick_rare_identifier( + "hybrid_search_async reciprocal_rank_fusion query", df.get) is None + + +def test_single_term_query_no_pick(): + assert _pick_rare_identifier("hybrid_search_async", {"hybrid_search_async": 1}.get) is None + + +# --- D2 exact gate (spec 2) --- +def test_exact_match_boosts_code_chunk(): + pool = [_chunk("other_func", score=9.0), + _chunk("hybrid_search_async", score=1.0)] + out = _boost_rare_identifier(pool, "hybrid_search_async") + assert out[0]["metadata"].get("symbol_name") == "hybrid_search_async" + assert out[0]["identifier_boost"] is True + assert out[0]["final_score"] == 1.0 * 100.0 # proven x100 convention + + +def test_substring_never_boosts(): + pool = [_chunk("my_hybrid_search_async_wrapper", score=9.0)] + out = _boost_rare_identifier(pool, "hybrid_search_async") + assert all("identifier_boost" not in r for r in out) + + +def test_doc_chunk_citing_identifier_not_boosted(): + """Negative control: doc chunk citing the identifier is NOT boosted.""" + pool = [_doc_chunk("hybrid_search_async"), + _chunk("unrelated", score=1.0)] + out = _boost_rare_identifier(pool, "hybrid_search_async") + assert all("identifier_boost" not in r for r in out) + assert out[0]["metadata"]["file"] == "KNOWN_ISSUES.md" # order untouched + + +def test_identifier_exact_match_helper(): + assert _identifier_exact_match(_chunk("hybrid_search_async"), "hybrid_search_async") + assert not _identifier_exact_match( + _chunk("my_hybrid_search_async_wrapper"), "hybrid_search_async") + assert not _identifier_exact_match( + _doc_chunk("hybrid_search_async"), "hybrid_search_async") + + +# --- D3 pre-rerank pool (spec 3) --- +def test_pre_rerank_pool_assertion_gold_survives(): + """Gold is deep in the pool pre-boost; O1 brings it to front pre-reranker.""" + pool = [_chunk(f"noise_{i}", score=10.0 - i) for i in range(8)] + gold = _chunk("hybrid_search_async", score=0.5) + pool.append(gold) + assert gold in pool # gold in pool PRE-boost (acceptance criterion) + out = _boost_rare_identifier(pool, "hybrid_search_async") + assert out[0] is gold + + +def test_searcher_method_end_to_end_with_fake_bm25(): + s = Searcher(MagicMock(), MagicMock()) + s._build_bm25_index = lambda: None # avoid real index build + # Fake BM25 stats: candidate rare, rest common. + s._bm25 = { + "a.py:0": {"hybrid_search_async": 1.0, "query": 1.0}, + "b.py:0": {"query": 1.0, "search": 1.0}, + "c.py:0": {"query": 1.0, "search": 1.0, "bm25": 1.0}, + } + pool = [_chunk("search", score=9.0), + _chunk("hybrid_search_async", score=1.0)] + out = s._apply_o1_identifier_boost( + pool, "hybrid_search_async search query") + assert out[0]["metadata"].get("symbol_name") == "hybrid_search_async" + assert out[0]["identifier_boost"] is True + + +def test_searcher_method_no_bm25_degrades_cleanly(): + s = Searcher(MagicMock(), MagicMock()) + s._build_bm25_index = lambda: None + s._bm25 = {} + pool = [_chunk("hybrid_search_async", score=1.0)] + assert s._apply_o1_identifier_boost(pool, "hybrid_search_async query") == pool