Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions KNOWN_ISSUES.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,11 @@

---

## 2026-09-28 — Ретриевер-замеры без сброса реранкер-кэша недействительны (Open)

- **Правило:** все retriever-замеры и A/B-тесты — только в свежем процессе либо с явным сбросом реранкер-кэша (`Searcher._reranker_cache.clear()`). Ключ кэша включает текст запроса (engine.py:1646): повтор того же запроса в том же процессе отдаёт закэшированные скоры, а не измеряет код.
- **Эвристика void-замера:** wall <2s на `hybrid_search_async` при ожидании полного пайплайна (embed+BM25+FTS+rerank) = подозрение на cache hit; сверяться с `Searcher._last_rerank_timing` (пусто = реранкер не работал). Холодный FTS-билд (~2.5s) — обратная ловушка: ПЕРВЫЙ замер в свежем процессе молча теряет FTS-тир (2s `wait_for`), нужен discarded warm-up на чужом запросе.
- **Статус:** 🟡 Open (процедурное правило; guard-скрипт `scripts/o1_holdout_gate.py` — fresh-process + warm-up + void-флаг).

**21 entries** — compressed per §4.8 R3 (conclusion-first; dedup 2026-09-08, 2026-09-21). Closed entries moved to docs/archive/KNOWN_ISSUES_2026_09.md on 2026-09-27 (R1 size guard; second batch on merge experiment/4a-unit-of-return).

Expand Down
144 changes: 144 additions & 0 deletions scripts/o1_holdout_gate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,144 @@
#!/usr/bin/env python3
"""O1 holdout gate (live, fresh-process): P2 + H1-H12 + N/doc controls.

Usage:
python scripts/o1_holdout_gate.py [--project D:/Project/MSCodeBase]

Each query runs hybrid_search_async(limit=5) in THIS fresh process
(reranker cache starts empty -> no cache-hit void measurements).
Prints a result table; exit 1 on gate failure (see PASS RULES below).

Holdout set note: no canonical H1-H12 exists on main (verified 2026-09-28:
grep finds only unrelated H-labels in diary/archive). H1-H12 below are
O1-holdout-v1, defined HERE (multi-token identifier queries with known
target files, disjoint from the frozen eval-16 calibration use).
"""
import argparse
import asyncio
import sys
import time
import traceback
from pathlib import Path

if sys.stdout.encoding and sys.stdout.encoding.lower() != "utf-8":
try:
sys.stdout.reconfigure(encoding="utf-8")
except Exception: # noqa: BLE001 — encoding guard must never fail startup
pass

ROOT = Path(__file__).resolve().parent.parent
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))

P2 = {"id": "P2", "query": "hybrid_search_async reciprocal_rank_fusion FTS5 BM25",
"target": "src/core/search/engine.py", "expect_rank": 1}

HOLDOUT = [
{"id": "H1", "query": "hybrid_search_async BM25 vector FTS5 ranking",
"target": "src/core/search/engine.py"},
{"id": "H2", "query": "BM25 common term index search query ranking",
"target": None, "expect_no_boost": True}, # collision control
{"id": "H3", "query": "ArtifactGC 30d projects 90d telemetry retention",
"target": "src/core/artifact_gc.py"},
{"id": "H4", "query": "resolve_ubatch embed 512 rerank 1024 batch",
"target": "src/providers/reranker/llama_install.py"},
{"id": "H5", "query": "_is_pid_alive OpenProcess SYNCHRONIZE stale PID guard",
"target": "src/providers/reranker/llama_runner.py"},
{"id": "H6", "query": "add_node add_edge SQLite transaction per node PropertyGraph",
"target": "src/core/graph.py"},
{"id": "H7", "query": "LLAMA_EMBED_MAX_TOKENS truncates 480 e5-small context",
"target": "src/providers/embedder/remote_embedder.py"},
{"id": "H8", "query": "json_group_array collect Cypher cypher_sql translation",
"target": "src/core/search/cypher_sql.py"},
{"id": "H9", "query": "bootstrap_tests TESTS edges verifying tests far better",
"target": "src/core/bootstrap_tests.py"},
{"id": "H10", "query": "to_pandas columns lancedb FreshnessChecker broken dead code",
"target": "src/core/indexing/freshness.py"},
{"id": "H11", "query": "CodeParser tree_sitter parse_file AST chunks walk",
"target": "src/core/indexing/parser.py"},
{"id": "H12", "query": "_cleanup_old_projects 30d retention_policy ArtifactGC",
"target": "src/core/artifact_gc.py"},
{"id": "N1", "query": "quantum computing configuration",
"target": None, "expect_no_boost": True}, # out-of-domain negative
{"id": "DOC", "query": "begin_write _write_lock RLock reindex freeze server",
"target": "src/core/indexing/db_writer.py", "expect_no_doc_boost": True},
]


def build_searcher(project: Path):
from src.core.di_container import IndexerFactoryKey, create_service_collection

services = create_service_collection(project)
factory = services.resolve(IndexerFactoryKey)
indexer = factory(project)
return indexer.searcher


def rank_of(results, target):
for i, r in enumerate(results, 1):
if (r.get("metadata") or {}).get("file") == target:
return i
return None


def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--project", default=str(ROOT))
args = ap.parse_args()
project = Path(args.project)
try:
import subprocess # noqa: PLC0415
rev = subprocess.run(["git", "rev-parse", "--short", "HEAD"],
cwd=str(ROOT), capture_output=True,
timeout=10).stdout.decode().strip()
except Exception: # noqa: BLE001 — git metadata is best-effort diagnostics
rev = "unknown"
print(f"O1 holdout gate | rev={rev} | project={project} | fresh process")
searcher = build_searcher(project)
# Warm-up (discarded): cold FTS5 to_pandas build exceeds the 2s tier budget
# on main and would silently drop the FTS tier for the FIRST measured query
# (known issue, PR52 territory). Warm-up uses a disjoint query -> no
# reranker-cache contamination (cache key includes the query text).
asyncio.run(searcher.hybrid_search_async("warmup cold start primer", limit=3))
rows = []
for case in [P2, *HOLDOUT]:
t0 = time.perf_counter()
results = asyncio.run(searcher.hybrid_search_async(case["query"], limit=5))
wall = time.perf_counter() - t0
boosted = [r for r in results if r.get("identifier_boost")]
doc_boosted = [r for r in boosted
if str((r.get("metadata") or {}).get("file", ""))
.lower().endswith((".md", ".markdown", ".rst", ".txt", ".ipynb"))]
rank = rank_of(results, case["target"]) if case.get("target") else None
void = wall < 2.0 and bool(getattr(searcher, "_last_rerank_timing", None) == {})
rows.append({"id": case["id"], "rank": rank, "target": case.get("target"),
"n_boost": len(boosted), "n_doc_boost": len(doc_boosted),
"wall": round(wall, 2), "void": void})
print(f"{case['id']:>3} rank={rank} target={case.get('target')} "
f"boost={len(boosted)} doc_boost={len(doc_boosted)} wall={wall:.2f}s")
print("---")
fails = []
p2 = rows[0]
if p2["rank"] != 1:
fails.append(f"P2 rank={p2['rank']} (expected 1)")
for row in rows:
case = next(c for c in [P2, *HOLDOUT] if c["id"] == row["id"])
if case.get("expect_no_boost") and row["n_boost"]:
fails.append(f"{row['id']} unexpectedly boosted")
if case.get("expect_no_doc_boost") and row["n_doc_boost"]:
fails.append(f"{row['id']} doc chunk boosted")
if row["void"]:
fails.append(f"{row['id']} reranker-cache void (wall<2s, no timing)")
if fails:
print("GATE FAIL: " + "; ".join(fails))
return 1
print("GATE PASS")
return 0


if __name__ == "__main__":
try:
sys.exit(main())
except Exception: # noqa: BLE001 — gate must report, never crash silently
traceback.print_exc()
sys.exit(1)
148 changes: 148 additions & 0 deletions src/core/search/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,7 @@
_extract_key_terms,
_extract_symbol_name,
_filter_by_time,
_tokenize,
)

_sync_executor = concurrent.futures.ThreadPoolExecutor(
Expand Down Expand Up @@ -135,6 +136,112 @@ def _boost_exact_name_matches(results: List[dict], query: str) -> List[dict]:
return results


# O1 (2026-09-28): identifier boost for MULTI-token queries at pre-rerank pool.
# Single-identifier queries are served by _boost_exact_name_matches (+ graph
# short-circuit); multi-token queries ("hybrid_search_async reciprocal_rank_fusion
# FTS5 BM25") never reach it (_IDENTIFIER_QUERY_RE rejects spaces), so a rare
# identifier inside them drowned in multi-tier RRF noise. Red-team defenses
# (mandatory acceptance criteria):
# D1 rarity gate — candidate only if it is the STRICTLY rarest query term
# (df from existing BM25 stats, no new index structures); a common-term
# collision (H2-style: `BM25` frequent in index) never boosts.
# D2 exact gate — full-token equality token == symbol_name, never substring;
# _is_doc_chunk exclusion inherited (docs never boosted).
# D3 pre-rerank application — runs on the pre-rerank pool (before
# _apply_multi_reranker_async), never post-cut; the x100 only guarantees
# pool survival past the reranker's top_n cut (reranker re-sorts itself).
# Shape classes mirror CodeJury _identifiers: --flag, UPPER, _ in name, CamelCase.
_IDENTIFIER_TOKEN_RES = (
re.compile(r"^--[A-Za-z][A-Za-z0-9_-]*$"), # --flag
re.compile(r"^[A-Z][A-Z0-9_]{1,63}$"), # CONSTANT / BM25 / FTS5
re.compile(r"^[A-Za-z_][A-Za-z0-9_]*_[A-Za-z0-9_]+$"), # snake_case
re.compile(r"^[a-z]+[A-Z][A-Za-z0-9]*$"), # camelCase
re.compile(r"^[A-Z][a-z0-9]+[A-Z][A-Za-z0-9]*$"), # PascalCase
)

# Proven x100 factor convention (cf. _boost_exact_name_matches).
_O1_BOOST_FACTOR = 100.0

# Same split pattern as Searcher._tokenizer_re: BM25 terms and O1 query terms
# must share tokenization, otherwise df comparison is meaningless.
_O1_TOKENIZER_RE = re.compile(r"\W+")


def _extract_identifier_tokens(query: str) -> List[str]:
"""Identifier-shaped raw tokens of the query (O1/D1 shape classes)."""
toks = re.findall(
r"--[A-Za-z][A-Za-z0-9_-]*|[A-Za-z_][A-Za-z0-9_]*", query or ""
)
return [t for t in toks if any(rx.match(t) for rx in _IDENTIFIER_TOKEN_RES)]


def _identifier_norm(tok: str) -> str:
"""Normalized form comparable with BM25 terms (lowercase, no leading dashes)."""
return tok.lower().lstrip("-")


def _pick_rare_identifier(query: str, df_of) -> Optional[str]:
"""O1 candidate: identifier-shaped token that is the STRICTLY rarest query term.

Args:
query: raw multi-token query.
df_of: callable term -> int, document frequency from existing BM25 stats.

Returns:
Normalized candidate or None. None covers: no identifier-shaped token,
single-term query (served by _boost_exact_name_matches), tie for rarest,
and H2-style collision (identifier-shaped but common term).
"""
idents = _extract_identifier_tokens(query)
if not idents:
return None
qtokens = _tokenize(query, _O1_TOKENIZER_RE)
uniq = set(qtokens)
if len(uniq) < 2:
return None
df_cache: Dict[str, int] = {}

def df(t: str) -> int:
if t not in df_cache:
try:
df_cache[t] = int(df_of(t) or 0)
except Exception: # noqa: BLE001 — broken stats never break search
df_cache[t] = 0
return df_cache[t]

for raw in idents:
norm = _identifier_norm(raw)
others = uniq - {norm}
if not others:
continue
if df(norm) < min(df(t) for t in others):
return norm
return None


def _identifier_exact_match(r: Dict, candidate_norm: str) -> bool:
"""True only on full-token equality symbol == candidate (O1/D2, never substring)."""
if _is_doc_chunk(r.get("metadata") or {}):
return False
meta = r.get("metadata") or {}
text = str(r.get("text", "") or "")
sym = str(meta.get("symbol_name", "") or "")
if not sym:
sym = _extract_symbol_name(text)
return bool(sym) and sym.lower() == candidate_norm


def _boost_rare_identifier(pool: List[dict], candidate_norm: str) -> List[dict]:
"""Applies the O1 x100 boost to exact symbol matches inside the pool."""
for r in pool:
if _identifier_exact_match(r, candidate_norm):
r["final_score"] = (r.get("final_score", 0) or 0) * _O1_BOOST_FACTOR
r["identifier_boost"] = True
# Stable: boosted first (input order), rest follow (input order).
pool.sort(key=lambda r: 0 if r.get("identifier_boost") else 1)
return pool


def _prepend_code_name_matches(
results: List[dict], pool: Optional[List[dict]], query: str, limit: int
) -> List[dict]:
Expand Down Expand Up @@ -517,6 +624,38 @@ def hybrid_search(
)
)

def _apply_o1_identifier_boost(
self, pool: List[dict], query: str
) -> List[dict]:
"""O1: rare-identifier x100 boost on the pre-rerank pool (defenses D1-D3).

Builds df from the EXISTING BM25 stats (no new index structures),
picks the strictly-rarest identifier-shaped query token and boosts
exact symbol matches. Any failure degrades to unboosted pool.
"""
if not pool:
return pool
try:
self._build_bm25_index()
except Exception: # noqa: BLE001 — degraded BM25 never breaks search
return pool
bm25 = getattr(self, "_bm25", None) or {}
if not bm25:
return pool

def df_of(term: str) -> int:
n = 0
for doc_terms in bm25.values():
if term in doc_terms:
n += 1
return n

candidate = _pick_rare_identifier(query, df_of)
if not candidate:
return pool
logger.debug(f"[O1] rare-identifier boost: '{candidate}' x{_O1_BOOST_FACTOR:g}")
return _boost_rare_identifier(pool, candidate)

async def hybrid_search_async(
self,
query: str,
Expand Down Expand Up @@ -757,6 +896,15 @@ async def hybrid_search_async(
if tracer and _mmr_before:
tracer.record_mmr(_mmr_before, pre_rerank_results, lambda_param=0.6)

# === O1 (2026-09-28): rare-identifier boost at PRE-RERANK pool (D3) ===
# Runs here — after sort+cut/MMR, before the reranker — never post-cut:
# the x100 guarantees a rare exact symbol match survives the reranker's
# top_n cut; the reranker itself still re-sorts by its own scores.
# (On PR52 this site hosts anchor_tier_winners; O1 composes after it.)
pre_rerank_results = self._apply_o1_identifier_boost(
pre_rerank_results, query
)

# Мульти-провайдерный реранкинг (Ollama / LM Studio) — опциональный
# Реранкер перезаписывает final_score своими семантическими весами
_pre_rerank = list(pre_rerank_results) if tracer else None
Expand Down
Loading
Loading