From 93bb342247a7c1d79ae75c6a5c46945f1342909e Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Mon, 28 Sep 2026 19:54:25 +0300 Subject: [PATCH 1/7] fix(search): cold FTS build out of 2s budget plus ext-dir off-by-one Cold to_pandas build measured 2.47s > 2.0s wait_for: first search in a fresh process silently dropped the FTS tier (flaky A/B 18/20 vs 8/20). Build hoisted out of timeout (idempotent, locked); 2s kept for search. _get_ext_dir took 3 parents instead of 4: dev-mode branch dead, models resolved to empty data_root. Embedder GGUF copied to git-ignored models/. --- KNOWN_ISSUES.md | 7 +++++++ src/core/search/engine.py | 10 ++++++---- src/providers/reranker/llama_install.py | 7 +++++-- 3 files changed, 18 insertions(+), 6 deletions(-) diff --git a/KNOWN_ISSUES.md b/KNOWN_ISSUES.md index c7b8c7dd..8d05771e 100644 --- a/KNOWN_ISSUES.md +++ b/KNOWN_ISSUES.md @@ -37,6 +37,13 @@ **24 entries** — compressed per §4.8 R3 (conclusion-first; dedup 2026-09-08, 2026-09-21). Closed entries moved to docs/archive/KNOWN_ISSUES_2026_09.md on 2026-09-27 (R1 size guard; second batch on merge experiment/4a-unit-of-return). +## 2026-09-28 — Холодный FTS-билд превышал 2s-бюджет и молча выпадал (Fixed) + _get_ext_dir указывал в src/ (Fixed) + +- **FTS (c, flaky A/B):** замер — холодный `to_pandas`-билд всего индекса = **2.47s > 2.0s** `wait_for` в `engine.py:671`. Первый поиск в свежем процессе молча терял FTS-тир → пилот 18/20 vs 8/20 на тех же запросах. **Fix:** build вынесен из-под таймаута (идемпотентен, double-checked lock), 2s остались только на сам поиск (~0.05s). Guard `test_fts5_timeout_does_not_break_search` зелёный. +- **llama-пути (b):** `llama_install.py:_get_ext_dir` брал 3 `parent` от `__file__` вместо 4 → указывал в `src/`, ветка «режим разработки» была мёртвой, модели резолвились в пустой `%LOCALAPPDATA%/mscodebase/models`. На вопрос «падает или не успевает»: после простоя restart **пытается** (`idle-unload recovery`), но падал по отсутствию файлов, не по таймингу. **Fix:** off-by-one исправлен + `multilingual-e5-small-Q8_0.gguf` (132MB) докопирован из расширения в `models/` (git-ignored). Live-check `smoke_e2e.py`: **SMOKE E2E PASSED** (embed dim=384, rerank top=1, поиск по индексу). +- **Побочно (Verified, не чинено — решение владельца):** холодный топ захламлён артефактами (`judged_raw*.json`, `work/ctx_*.txt` в выдаче) — живое подтверждение индексного мусора (P2-смежное). Чистка индекса сменит ретрив-базисы. +- **Статус:** ✅ Fixed (пути + FTS-холод). + ## 2026-09-27 — Import-time os.environ mutation in scripts breaks xdist workers (Fixed) - **Симптом:** 6 plugin-тестов (`test_plugins_subprocess/registry`) падали под `-n auto` с `ModuleNotFoundError: No module named 'src'` в runner-subprocess, серийно (`-n0`) — зелёные. diff --git a/src/core/search/engine.py b/src/core/search/engine.py index 2323cbc7..0cccf132 100644 --- a/src/core/search/engine.py +++ b/src/core/search/engine.py @@ -962,10 +962,12 @@ async def hybrid_search_async( except Exception as e: logger.warning(f"Не удалось выполнить dense поиск: {e}") - # FTS5 (full-text) — параллельно, с защитой от таймаута. - # _fts5_search делает lazy build (to_pandas на весь индекс, ~0.5s на - # первом вызове). Чтобы не усугублять 15s-лимит search_code, оборачиваем - # в wait_for(2s): при превышении — degraded ([]), основной поиск жив. + # FTS5 (full-text) — build вне таймаута, поиск под wait_for(2s). + # Замер 2026-09-28: холодный build (to_pandas всего индекса) = 2.47s > + # 2.0s — первый поиск в свежем процессе МОЛЧА терял FTS-тир + # (flaky A/B: пилот 18/20 vs 8/20 на тех же запросах). Build идемпотентен + # (double-checked lock в _build_fts5_index), поиск — быстрый (~0.05s). + await asyncio.to_thread(self._build_fts5_index) try: fts5_raw = await asyncio.wait_for( self._fts5_search_async(query, limit=raw_limit * 2), diff --git a/src/providers/reranker/llama_install.py b/src/providers/reranker/llama_install.py index d39db3a3..55e6ddb3 100644 --- a/src/providers/reranker/llama_install.py +++ b/src/providers/reranker/llama_install.py @@ -307,8 +307,11 @@ def _get_ext_dir() -> Path: p = Path(sys.executable).resolve().parent.parent.parent if (p / "src" / "main.py").exists() and (p / "__mscodebase_ext__.marker").exists(): return p - # Режим разработки (исходники с маркером расширения в корне) - p = Path(__file__).resolve().parent.parent.parent + # Режим разработки (исходники с маркером расширения в корне). + # __file__ = /src/providers/reranker/llama_install.py → 4 уровня + # до корня (было 3 — указывало на src/, ветка была мёртвой, и модели + # резолвились в пустой data_root; найдено 2026-09-28 по отсутствию GGUF). + p = Path(__file__).resolve().parent.parent.parent.parent if (p / "src" / "main.py").exists() and (p / "__mscodebase_ext__.marker").exists(): return p # Установленный пакет (pip/uvx) или неизвестный контекст: единый data root From c4c9296fdc707bf5b9f3a79a74c052edd081ebf7 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Mon, 28 Sep 2026 20:02:37 +0300 Subject: [PATCH 2/7] feat(index): exclude experiment outputs from index, keep git tracking 3127/15426 chunks (20.3 pct) were experiments slash results|work dumps (ctx files up to 329 chunks each) choking lexical search (P2) and cold FTS build (2.47s). New SystemArtifacts.is_experiment_output layer wired into FileGuard; frozen inputs and experiment sources stay indexed. Takes effect on full reindex. Guard: test_experiment_output_guard. --- src/core/indexing/file_guard.py | 6 ++++ src/core/system_artifacts.py | 32 +++++++++++++++++-- tests/test_experiment_output_guard.py | 46 +++++++++++++++++++++++++++ 3 files changed, 82 insertions(+), 2 deletions(-) create mode 100644 tests/test_experiment_output_guard.py diff --git a/src/core/indexing/file_guard.py b/src/core/indexing/file_guard.py index 22044ea5..02cf3c53 100644 --- a/src/core/indexing/file_guard.py +++ b/src/core/indexing/file_guard.py @@ -138,6 +138,12 @@ def is_safe_to_index(self, file_path: Path) -> bool: logger.debug(f"[FILEGUARD SKIP] System directory: {file_path}") return False + # Выводы экспериментов (experiments/**/results|work) — в индекс не берём. + # Git-трекинг не трогаем: frozen/results обязаны жить в репо (§17). + if SystemArtifacts.is_experiment_output(file_path): + logger.debug(f"[FILEGUARD SKIP] Experiment output: {file_path}") + return False + # Проверка .gitignore (Требует POSIX путей) if self._gitignore_patterns: try: diff --git a/src/core/system_artifacts.py b/src/core/system_artifacts.py index bfd97809..da6e0093 100644 --- a/src/core/system_artifacts.py +++ b/src/core/system_artifacts.py @@ -101,9 +101,20 @@ # ══════════════════════════════════════════════════════════════ -# Public API +# Layer 4: Experiment Output Guard — выводы экспериментов # ══════════════════════════════════════════════════════════════ +# Замер 2026-09-28: 3127/15426 чанков индекса (20.3%) — мусор из +# experiments/**/results|work (ctx-дампы по 329 чанков, judged_raw.json — +# 249). Душит лексику (кейс P2) и раздувает холодный FTS-билд (2.47s). +# Git-трекинг НЕ трогаем (§17: frozen/results обязаны жить в репо) — +# исключаем только из ИНДЕКСА. Исходники экспериментов (*.py) индексируются. +_EXPERIMENT_ROOT = "experiments" +_EXPERIMENT_OUTPUT_DIRS = frozenset({"results", "work"}) + + +# ─── Public API ───────────────────────────────────────────── + class SystemArtifacts: """Единый источник правды о системных файлах проекта. @@ -197,7 +208,24 @@ def is_feedback_risk(cls, path: Path) -> bool: name = path.name.lower() return name in _FEEDBACK_PATTERNS - # ─── Layer 4: Unified Check ───────────────────────────── + # ─── Layer 4: Experiment Output Guard ──────────────── + + @classmethod + def is_experiment_output(cls, path: Path) -> bool: + """Проверяет, является ли файл выводом эксперимента. + + experiments/**/results/** и experiments/**/work/** — сырьё прогонов + (ctx-дампы, judged_raw.json, work-файлы). В индекс не берём; + frozen/-входы и *.py-исходники — берём. + """ + parts = [p.lower() for p in Path(path).parts] + try: + i = parts.index(_EXPERIMENT_ROOT) + except ValueError: + return False + return any(p in _EXPERIMENT_OUTPUT_DIRS for p in parts[i + 1 :]) + + # ─── Layer 5: Unified Check ───────────────────────────── @classmethod def is_system_path(cls, path: Path) -> bool: diff --git a/tests/test_experiment_output_guard.py b/tests/test_experiment_output_guard.py new file mode 100644 index 00000000..69a21346 --- /dev/null +++ b/tests/test_experiment_output_guard.py @@ -0,0 +1,46 @@ +"""Guard: выводы экспериментов не индексируются (experiments/**/results|work). + +Замер 2026-09-28: 3127/15426 чанков (20.3%) — мусор (ctx-дампы, judged_raw). +Git-трекинг не трогаем (§17) — только индекс. +""" +from __future__ import annotations + +from pathlib import Path + +from src.core.indexing.file_guard import FileGuard +from src.core.system_artifacts import SystemArtifacts + + +def test_results_and_work_excluded(): + assert SystemArtifacts.is_experiment_output( + Path("experiments/4A_unit_of_return/results/f5judged/judged_raw.json") + ) + assert SystemArtifacts.is_experiment_output( + Path("experiments/4A_unit_of_return/results/f5judged/work/ctx_F5S-01_A.txt") + ) + assert SystemArtifacts.is_experiment_output( + Path("D:/Project/MSCodeBase/experiments/noderag/results/r.json") + ) + + +def test_sources_and_frozen_kept(): + assert not SystemArtifacts.is_experiment_output( + Path("experiments/4A_unit_of_return/run_experiment.py") + ) + assert not SystemArtifacts.is_experiment_output( + Path("experiments/4A_unit_of_return/frozen/f5/queries.jsonl") + ) + assert not SystemArtifacts.is_experiment_output(Path("src/core/search/engine.py")) + assert not SystemArtifacts.is_experiment_output(Path("docs/en/SEARCH_PIPELINE.md")) + + +def test_fileguard_skips_experiment_output(tmp_path): + guard = FileGuard(tmp_path) + out = tmp_path / "experiments" / "x" / "results" / "r.json" + out.parent.mkdir(parents=True) + out.write_text('{"a": 1}', encoding="utf-8") + assert guard.is_safe_to_index(out) is False + + src = tmp_path / "experiments" / "x" / "run.py" + src.write_text("x = 1\n", encoding="utf-8") + assert guard.is_safe_to_index(src) is True From 322765da1f920a9efd33b5472d2a44c8e10d12c6 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Tue, 29 Sep 2026 15:20:05 +0300 Subject: [PATCH 3/7] tool(index): one-time purge of experiment outputs from shared DB 772 files / 2152 chunks under experiments slash results|work removed via delete_file plus graph cleanup plus compaction (prune safety-guard at 50pct would refuse: garbage was 52.4pct of files). Verified 0 remaining. Dry-run by default. --- scripts/purge_experiment_outputs.py | 89 +++++++++++++++++++++++++++++ 1 file changed, 89 insertions(+) create mode 100644 scripts/purge_experiment_outputs.py diff --git a/scripts/purge_experiment_outputs.py b/scripts/purge_experiment_outputs.py new file mode 100644 index 00000000..037abd72 --- /dev/null +++ b/scripts/purge_experiment_outputs.py @@ -0,0 +1,89 @@ +#!/usr/bin/env python3 +"""One-time purge: удалить выводы экспериментов из индекса (2026-09-28). + +Замер: 828 файлов / 3127 чанков (20.3%) под experiments/**/results|work. +Штатный prune_deleted_files отказывает (safety-guard >50%), т.к. файлы +НА диске — их newly-excludes FileGuard. Поэтому точечное удаление по +тому же delete-механизму, guard безопасности не трогаем. + +Usage: + python scripts/purge_experiment_outputs.py # dry-run + python scripts/purge_experiment_outputs.py --apply # удалить + compaction +""" +from __future__ import annotations + +import argparse +import sys +import traceback +from pathlib import Path + +if sys.stdout is not None: + try: + sys.stdout.reconfigure(encoding="utf-8") + except Exception: + pass + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--apply", action="store_true") + args = ap.parse_args() + + from src.core.artifact_paths import get_db_path + from src.core.di_container import create_service_collection + from src.core.indexing.file_guard import FileGuard + from src.core.indexing.indexer import Indexer + from src.core.indexing.parser import CodeParser + from src.core.indexing.symbol_index import SymbolIndex + from src.core.system_artifacts import SystemArtifacts + from src.providers.embedder.remote_embedder import RemoteEmbedder + + services = create_service_collection(ROOT) + embedder = services.resolve(RemoteEmbedder) + indexer = Indexer(db_path=get_db_path(ROOT), embedder=embedder, file_guard=FileGuard(ROOT), + project_path=ROOT, parser=CodeParser(), symbol_index=SymbolIndex()) + table = indexer.table + rows = table.search().limit(100000).to_list() + files = {r["file_path"] for r in rows if "file_path" in r} + garbage = sorted(f for f in files if SystemArtifacts.is_experiment_output(Path(f))) + gset = set(garbage) + n_chunks = sum(1 for r in rows if r.get("file_path") in gset) + print(f"files in db: {len(files)}, garbage files: {len(garbage)}, garbage chunks: {n_chunks}") + if not args.apply: + print("dry-run: nothing deleted (use --apply)") + return 0 + + tbl = indexer.indexer_table if hasattr(indexer, "indexer_table") else indexer + ok, fail = 0, 0 + for i, fp in enumerate(garbage, 1): + try: + if tbl.delete_file(fp): + ok += 1 + else: + fail += 1 + pg = getattr(getattr(tbl, "_symbol_index", None), "graph", None) + if pg: + pg.remove_file(str(fp).replace("\\", "/")) + except Exception as e: # noqa: BLE001 - one bad file must not stop purge + print(f" FAIL {fp}: {e}") + fail += 1 + if i % 200 == 0: + print(f" ...{i}/{len(garbage)}") + try: + table.compact_files() + print("compaction done") + except Exception as e: # noqa: BLE001 + print(f"compaction skipped: {e}") + print(f"purged files: {ok}, failed: {fail}") + return 0 if fail == 0 else 1 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + traceback.print_exc() + raise SystemExit(1) From e5aafe3a87863899186ddd7dd04e1e2667c5d0d5 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Tue, 29 Sep 2026 15:42:13 +0300 Subject: [PATCH 4/7] exp(f5relang): clean-stack RU vs EN rerun plus queries-file flag RU B 26/80 (32.5pct) vs EN B 30/80 (37.5pct), same clean index, same stack: CIs overlap, no significant language effect. 6/16 queries flip per-language (all-or-nothing swaps, e.g. F5S-02 5-0): language changes WHICH queries pass, not how many. Harness gains --queries-file (variants live outside frozen/). --- .../results/f5relang/aggregate.json | 58 + .../results/f5relang/en/judged_raw.json | 676 +++ .../results/f5relang/en/work/cand_c00000.txt | 5 + .../results/f5relang/en/work/cand_c00001.txt | 5 + .../results/f5relang/en/work/cand_c00002.txt | 5 + .../results/f5relang/en/work/cand_c00003.txt | 5 + .../results/f5relang/en/work/cand_c00004.txt | 5 + .../results/f5relang/en/work/cand_c00005.txt | 5 + .../results/f5relang/en/work/cand_c00006.txt | 5 + .../results/f5relang/en/work/cand_c00007.txt | 5 + .../results/f5relang/en/work/cand_c00008.txt | 5 + .../results/f5relang/en/work/cand_c00009.txt | 5 + .../results/f5relang/en/work/cand_c00010.txt | 5 + .../results/f5relang/en/work/cand_c00011.txt | 5 + .../results/f5relang/en/work/cand_c00012.txt | 5 + .../results/f5relang/en/work/cand_c00013.txt | 5 + .../results/f5relang/en/work/cand_c00014.txt | 5 + .../results/f5relang/en/work/cand_c00015.txt | 5 + .../results/f5relang/en/work/ctx_c00000.txt | 123 + .../results/f5relang/en/work/ctx_c00001.txt | 139 + .../results/f5relang/en/work/ctx_c00002.txt | 323 ++ .../results/f5relang/en/work/ctx_c00003.txt | 209 + .../results/f5relang/en/work/ctx_c00004.txt | 946 +++++ .../results/f5relang/en/work/ctx_c00005.txt | 837 ++++ .../results/f5relang/en/work/ctx_c00006.txt | 3672 +++++++++++++++++ .../results/f5relang/en/work/ctx_c00007.txt | 23 + .../results/f5relang/en/work/ctx_c00008.txt | 797 ++++ .../results/f5relang/en/work/ctx_c00009.txt | 164 + .../results/f5relang/en/work/ctx_c00010.txt | 492 +++ .../results/f5relang/en/work/ctx_c00011.txt | 33 + .../results/f5relang/en/work/ctx_c00012.txt | 72 + .../results/f5relang/en/work/ctx_c00013.txt | 492 +++ .../results/f5relang/en/work/ctx_c00014.txt | 47 + .../results/f5relang/en/work/ctx_c00015.txt | 67 + .../results/f5relang/en/work/opencode.json | 24 + .../results/f5relang/queries_en.jsonl | 16 + .../results/f5relang/ru/judged_raw.json | 676 +++ .../results/f5relang/ru/work/cand_c00000.txt | 5 + .../results/f5relang/ru/work/cand_c00001.txt | 5 + .../results/f5relang/ru/work/cand_c00002.txt | 5 + .../results/f5relang/ru/work/cand_c00003.txt | 5 + .../results/f5relang/ru/work/cand_c00004.txt | 5 + .../results/f5relang/ru/work/cand_c00005.txt | 5 + .../results/f5relang/ru/work/cand_c00006.txt | 5 + .../results/f5relang/ru/work/cand_c00007.txt | 5 + .../results/f5relang/ru/work/cand_c00008.txt | 5 + .../results/f5relang/ru/work/cand_c00009.txt | 5 + .../results/f5relang/ru/work/cand_c00010.txt | 5 + .../results/f5relang/ru/work/cand_c00011.txt | 5 + .../results/f5relang/ru/work/cand_c00012.txt | 5 + .../results/f5relang/ru/work/cand_c00013.txt | 5 + .../results/f5relang/ru/work/cand_c00014.txt | 5 + .../results/f5relang/ru/work/cand_c00015.txt | 5 + .../results/f5relang/ru/work/ctx_c00000.txt | 277 ++ .../results/f5relang/ru/work/ctx_c00001.txt | 495 +++ .../results/f5relang/ru/work/ctx_c00002.txt | 323 ++ .../results/f5relang/ru/work/ctx_c00003.txt | 209 + .../results/f5relang/ru/work/ctx_c00004.txt | 333 ++ .../results/f5relang/ru/work/ctx_c00005.txt | 244 ++ .../results/f5relang/ru/work/ctx_c00006.txt | 438 ++ .../results/f5relang/ru/work/ctx_c00007.txt | 286 ++ .../results/f5relang/ru/work/ctx_c00008.txt | 96 + .../results/f5relang/ru/work/ctx_c00009.txt | 21 + .../results/f5relang/ru/work/ctx_c00010.txt | 70 + .../results/f5relang/ru/work/ctx_c00011.txt | 23 + .../results/f5relang/ru/work/ctx_c00012.txt | 72 + .../results/f5relang/ru/work/ctx_c00013.txt | 320 ++ .../results/f5relang/ru/work/ctx_c00014.txt | 21 + .../results/f5relang/ru/work/ctx_c00015.txt | 21 + .../results/f5relang/ru/work/opencode.json | 24 + scripts/f5_judged_run.py | 10 +- 71 files changed, 13326 insertions(+), 3 deletions(-) create mode 100644 experiments/4A_unit_of_return/results/f5relang/aggregate.json create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/judged_raw.json create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00000.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00001.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00002.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00003.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00004.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00005.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00006.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00007.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00008.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00009.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00010.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00011.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00012.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00013.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00014.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00015.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00000.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00001.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00002.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00003.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00004.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00005.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00006.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00007.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00008.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00009.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00010.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00011.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00012.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00013.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00014.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00015.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/en/work/opencode.json create mode 100644 experiments/4A_unit_of_return/results/f5relang/queries_en.jsonl create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/judged_raw.json create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00000.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00001.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00002.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00003.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00004.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00005.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00006.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00007.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00008.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00009.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00010.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00011.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00012.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00013.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00014.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/cand_c00015.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00000.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00001.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00002.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00003.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00004.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00005.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00006.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00007.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00008.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00009.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00010.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00011.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00012.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00013.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00014.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/ctx_c00015.txt create mode 100644 experiments/4A_unit_of_return/results/f5relang/ru/work/opencode.json diff --git a/experiments/4A_unit_of_return/results/f5relang/aggregate.json b/experiments/4A_unit_of_return/results/f5relang/aggregate.json new file mode 100644 index 00000000..12394a96 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/aggregate.json @@ -0,0 +1,58 @@ +{ + "ru": { + "correct": 26, + "n": 80, + "rate": 0.325, + "wilson95": [ + 0.232, + 0.434 + ], + "code_rate": 0.5, + "per_query": { + "F5S-01": 0, + "F5S-02": 5, + "F5S-03": 0, + "F5S-04": 0, + "F5S-05": 5, + "F5S-06": 5, + "F5S-07": 5, + "F5S-08": 0, + "F5S-09": 0, + "F5S-10": 0, + "F5S-11": 0, + "F5S-12": 0, + "F5S-13": 1, + "F5S-14": 0, + "F5S-15": 5, + "F5S-16": 0 + } + }, + "en": { + "correct": 30, + "n": 80, + "rate": 0.375, + "wilson95": [ + 0.277, + 0.485 + ], + "code_rate": 0.75, + "per_query": { + "F5S-01": 0, + "F5S-02": 0, + "F5S-03": 5, + "F5S-04": 5, + "F5S-05": 5, + "F5S-06": 5, + "F5S-07": 5, + "F5S-08": 5, + "F5S-09": 0, + "F5S-10": 0, + "F5S-11": 0, + "F5S-12": 0, + "F5S-13": 0, + "F5S-14": 0, + "F5S-15": 0, + "F5S-16": 0 + } + } +} \ No newline at end of file diff --git a/experiments/4A_unit_of_return/results/f5relang/en/judged_raw.json b/experiments/4A_unit_of_return/results/f5relang/en/judged_raw.json new file mode 100644 index 00000000..2b523e2c --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/judged_raw.json @@ -0,0 +1,676 @@ +{ + "config": { + "outdir": "experiments/4A_unit_of_return/results/f5relang/en", + "n": 16, + "ids": "", + "arms": "B", + "trials": 5, + "judge_repeats": 1, + "parallel": 4, + "reader_model": "opencode-go/longcat-2.0", + "judge_model": "opencode-go/qwen3.7-plus", + "timeout": 400, + "seed": 42, + "reader_instr_file": null, + "queries_file": "experiments/4A_unit_of_return/results/f5relang/queries_en.jsonl", + "dry_run": false + }, + "records": [ + { + "id": "F5S-01", + "population": "code", + "gold_file": "src/core/rate_limiter.py", + "reference": "Защита от race conditions: threading.Lock (НЕ asyncio.Lock) —", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-02", + "population": "code", + "gold_file": "src/core/redact.py", + "reference": "Ловит только ИЗВЕСТНЫЕ ФОРМЫ по структурному префиксу.", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-03", + "population": "code", + "gold_file": "src/core/gitignore_parser.py", + "reference": "# Используем 'gitignore' вместо 'gitwildmatch' (deprecated)", + "arms": { + "B": { + "answers": [ + "Because 'gitwildmatch' is deprecated. The code explicitly uses the 'gitignore' format instead, as noted in the docstring and inline comment.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The parser uses the `'gitignore'` format because `'gitwildmatch'` is deprecated. Switching to `'gitignore'` eliminates 1740 GitWildMatchPattern deprecation warnings in tests.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "Because `'gitwildmatch'` is deprecated — switching to `'gitignore'` eliminates 1740 `GitWildMatchPattern` deprecation warnings in tests and speeds up .gitignore checks during indexing.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The parser uses the 'gitignore' format because 'gitwildmatch' is deprecated — this avoids the deprecation warnings and is the non-deprecated alternative.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The `'gitwildmatch'` format is deprecated. Using the `'gitignore'` format eliminates 1740 GitWildMatchPattern deprecation warnings and avoids re-parsing `.gitignore` on every file.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "correct", + "correct", + "correct", + "correct", + "correct" + ], + "judge_texts": [ + [ + "{\"verdict\": \"correct\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-04", + "population": "code", + "gold_file": "src/core/embedder_lease.py", + "reference": "Fix: an indexing process creates a short-lived *lease* file refreshed on every", + "arms": { + "B": { + "answers": [ + "The problem is solved with a cross-process **lease file**: an indexing process creates/short-lived lease file (`embedder.lease`) refreshed on every progress tick, and the idle watchdog checks for a fresh lease before killing — if any recent lease exists (own or another process's reindex), it must NOT kill the embedder.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The indexing process creates a short-lived lease file refreshed on every progress tick, and the idle watchdog is prohibited from killing the embedder while any fresh lease exists.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "A short-lived *lease* file is created on every progress tick by an indexing process; the idle watchdog must not kill the embedder while any fresh lease exists (its own reindex or another process's).\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "A short-lived lease file is created by the indexing process and refreshed on every progress tick. The idle watchdog checks for this lease and must NOT kill the embedder while any fresh lease exists (from its own reindex or another process's).\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The fix uses a short-lived **lease file** that an indexing process refreshes on every progress tick. The idle watchdog is required to check this lease and must **NOT kill the embedder while any fresh lease exists** (its own reindex or another process's).\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "correct", + "correct", + "correct", + "correct", + "correct" + ], + "judge_texts": [ + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-05", + "population": "code", + "gold_file": "src/core/reindex_ledger.py", + "reference": "- MUST NOT raise, ever (a recording channel cannot break indexing).", + "arms": { + "B": { + "answers": [ + "The main constraint is that the module MUST NOT raise, ever — a recording channel cannot break indexing. It also must be append-only and use a single artifact path convention, but the non-negotiable rule is swallowing every error by design.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The module **MUST NOT raise, ever** — it swallows all errors by design, because a recording channel cannot be allowed to break indexing.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The module MUST NOT raise any exception — it is designed to swallow every error so that a recording channel can never break indexing.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The main constraint is that the module **MUST NOT raise, ever** — as stated in the design constraints, \"a recording channel cannot break indexing.\"\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The module **MUST NOT raise, ever** — it swallows every error by design so that the recording channel cannot break indexing.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "correct", + "correct", + "correct", + "correct", + "correct" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-06", + "population": "code", + "gold_file": "src/core/quiet_break_gate.py", + "reference": "(fail-open, причина логируется): инфраструктурный сбой не должен блокировать коммит.", + "arms": { + "B": { + "answers": [ + "On infrastructural git or graph failure, the gate returns status `\"unavailable\"` (or `\"empty\"`) with an empty findings list and **does not trigger** (fail-open): the failure is logged but does not block the commit, because an infrastructure outage should not halt development.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "On infrastructural git or graph failure, the gate returns status `\"unavailable\"` with empty findings and does **not** trigger (fail-open) — the reason is logged, but the commit is not blocked.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "On infrastructural git or graph failure, the gate returns status `\"unavailable\"` with empty findings and does **not** block the commit (fail-open). It never throws an exception outward, and the reason is logged.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "On infrastructural git or graph failure, the gate returns status `\"unavailable\"` with empty findings and does **not** trigger (fail-open), logging the reason. This ensures an infrastructural failure does not block the commit.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "On infrastructural failure (missing git, absent PropertyGraph, or unreachable DB), the gate returns status `\"unavailable\"` with empty `findings` and does **not** block the commit — it fails open, logging the reason. This ensures an infrastructure outage never halts development.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "correct", + "correct", + "correct", + "correct", + "correct" + ], + "judge_texts": [ + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-07", + "population": "code", + "gold_file": "src/providers/reranker/reranker_scoring.py", + "reference": "score: конечное число. NaN/Inf отбрасываются", + "arms": { + "B": { + "answers": [ + "`validate_scores` discards items with NaN/Inf scores (line 70-71: `if not math.isfinite(score): continue`). The docstring explains this is intentional — NaN/Inf represent an absence of assessment, and previously min/max clamping would silently convert NaN to 1.0 (maximum score for an unscored chunk).\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "`validate_scores` discards (filters out) items with NaN or Inf scores — they are not included in the validated list. The code at line 70-71 checks `math.isfinite(score)` and skips non-finite values, preventing NaN from silently becoming 1.0 via clamp.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "validate_scores discards elements where the score is NaN or Inf — they fail the `math.isfinite()` check and are excluded from the validated list entirely, rather than being clamped to 1.0 or 0.0. This is fail-safe: a missing score is preferred over a silently rearranged one.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "`validate_scores` discards items with NaN or Inf scores — they are filtered out via `if not math.isfinite(score): continue`. The comment explains that NaN/Inf would otherwise pass `isinstance` checks and silently become 1.0 after min/max clamping, giving unscored chunks a maximum score. This is fail-safe: a lost score is better than a silently reordered one.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "`validate_scores` discards items with NaN or Inf scores — they are filtered out via `math.isfinite(score)` check (line 70-71), not clamped. The comment explains NaN/Inf would otherwise pass `isinstance` checks and silently become 1.0/0.0.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "correct", + "correct", + "correct", + "correct", + "correct" + ], + "judge_texts": [ + [ + "```json\n{\n \"verdict\": \"correct\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"correct\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-08", + "population": "code", + "gold_file": "src/core/error_envelope.py", + "reference": "``next_action`` hint so an agent can branch and recover without guessing", + "arms": { + "B": { + "answers": [ + "The `next_action` field gives a concrete, single-step hint (\"what to do about it\") so an agent can branch and recover from a tool failure without guessing — agents branch on `code`, while humans read `next_action`.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The `next_action` field exists so an agent can branch and recover from a tool failure without guessing — each hint is concrete and single-step. Agents branch on the error code, while humans read `next_action` to understand what to do next.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The `next_action` field provides a concrete, single-step hint that tells an agent how to branch and recover after a tool failure—so it doesn't have to guess what to do next.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The `next_action` field gives an agent a concrete, single-step hint so it can branch and recover from a tool failure without guessing what to do next. Each error code maps to a specific playbook instruction (e.g., fix arguments, trigger a reindex, or back off), and agents branch on `code` while humans read `next_action`.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The `next_action` field exists so an agent can branch and recover from tool failures without guessing. Agents branch on the stable `code`, while humans and agents read the concrete, single-step `next_action` hint to know exactly what to do next.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "correct", + "correct", + "correct", + "correct", + "correct" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"correct\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-09", + "population": "prose", + "gold_file": "docs/en/GRACEFUL_DEGRADATION.md", + "reference": "it **degrades gracefully** through 6 levels,", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n→ Skill \"agentic-eval-lab\"\r\n", + "I don't know. The context only describes one degradation level (falling back to temporary storage when `safe_mkdir` fails with `ArtifactStorageError`), but it says nothing about external providers.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"incorrect\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-10", + "population": "prose", + "gold_file": "docs/en/SEARCH_PIPELINE.md", + "reference": "Ranks are computed **separately** for each channel, starting from 1.", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-11", + "population": "prose", + "gold_file": "docs/TRUST_BOUNDARY.md", + "reference": "**Принцип: содержимое репозитория — это ДАННЫЕ, а не код.**", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-12", + "population": "prose", + "gold_file": "docs/en/TELEMETRY.md", + "reference": "Metrics persist across MCP server restarts — `load_metrics()` reads the saved JSON on startup.", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\n \"verdict\": \"incorrect\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-13", + "population": "prose", + "gold_file": "docs/adr/0003-verify-on-read.md", + "reference": "вердикт **INCONCLUSIVE, никогда VERIFIED** (fail-closed:", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-14", + "population": "prose", + "gold_file": "docs/en/HANDFOFF.md", + "reference": "**Solution:** read Zed's SQLite database directly:", + "arms": { + "B": { + "answers": [ + "The MCP server detects the project via the `$ZED_WORKTREE_ROOT` environment variable, which Zed resolves at launch time to the open project's worktree root directory. The server entry configures `PROJECT_PATH = \"$ZED_WORKTREE_ROOT\"` in its `env` dict, so Zed substitutes the actual path when starting the process.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The MCP server uses the `$ZED_WORKTREE_ROOT` environment variable (set as `PROJECT_PATH` at line 308), which Zed automatically injects when launching the context server — it resolves to the root of the currently open project/worktree, so no manual environment variables need to be passed.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The MCP server reads the `PROJECT_PATH` environment variable, which is set to `\"$ZED_WORKTREE_ROOT\"` — Zed automatically substitutes this token with the path to the currently open project's worktree root.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The MCP server detects the project via the `PROJECT_PATH` environment variable, which is set to `$ZED_WORKTREE_ROOT` (line 308 in `zed_config.py`). This is automatically injected by Zed itself, representing the root of the currently open worktree, so no manual environment variables need to be passed.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "The server entry sets `env[\"PROJECT_PATH\"] = \"$ZED_WORKTREE_ROOT\"` — Zed automatically injects this variable pointing to the open project's root, so the MCP server reads it at startup (see `_make_server_entry` at line 308).\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-15", + "population": "prose", + "gold_file": "docs/en/ZED_WINDOWS_QUIRKS.md", + "reference": "**Status:** ⚠️ Always `` on Windows (Zed bug #36019)", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n✱ Grep \"ZED_WORKTREE_ROOT\" in . · 5 matches\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\n \"verdict\": \"incorrect\"\n}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + }, + { + "id": "F5S-16", + "population": "prose", + "gold_file": "docs/en/SYSTEM_REQUIREMENTS.md", + "reference": "Moving to llama.cpp GGUF reduced MCP memory by 88% and total by 60%.", + "arms": { + "B": { + "answers": [ + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n", + "I don't know.\r\n\n\r\n> build · longcat-2.0\r\n\r\n" + ], + "verdicts": [ + "incorrect", + "incorrect", + "incorrect", + "incorrect", + "incorrect" + ], + "judge_texts": [ + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "{\"verdict\": \"incorrect\"}\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ], + [ + "```json\n{\"verdict\": \"incorrect\"}\n```\r\n\n\r\n> build · qwen3.7-plus\r\n\r\n" + ] + ] + } + } + } + ] +} \ No newline at end of file diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00000.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00000.txt new file mode 100644 index 00000000..99acb11e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00000.txt @@ -0,0 +1,5 @@ +The `next_action` field exists so an agent can branch and recover from tool failures without guessing. Agents branch on the stable `code`, while humans and agents read the concrete, single-step `next_action` hint to know exactly what to do next. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00001.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00001.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00001.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00002.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00002.txt new file mode 100644 index 00000000..f016cf98 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00002.txt @@ -0,0 +1,5 @@ +On infrastructural failure (missing git, absent PropertyGraph, or unreachable DB), the gate returns status `"unavailable"` with empty `findings` and does **not** block the commit — it fails open, logging the reason. This ensures an infrastructure outage never halts development. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00003.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00003.txt new file mode 100644 index 00000000..97da7b4e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00003.txt @@ -0,0 +1,5 @@ +`validate_scores` discards items with NaN or Inf scores — they are filtered out via `math.isfinite(score)` check (line 70-71), not clamped. The comment explains NaN/Inf would otherwise pass `isinstance` checks and silently become 1.0/0.0. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00004.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00004.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00004.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00005.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00005.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00005.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00006.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00006.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00006.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00007.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00007.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00007.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00008.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00008.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00008.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00009.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00009.txt new file mode 100644 index 00000000..c9fefdc1 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00009.txt @@ -0,0 +1,5 @@ +The `'gitwildmatch'` format is deprecated. Using the `'gitignore'` format eliminates 1740 GitWildMatchPattern deprecation warnings and avoids re-parsing `.gitignore` on every file. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00010.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00010.txt new file mode 100644 index 00000000..fb04f40c --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00010.txt @@ -0,0 +1,5 @@ +The server entry sets `env["PROJECT_PATH"] = "$ZED_WORKTREE_ROOT"` — Zed automatically injects this variable pointing to the open project's root, so the MCP server reads it at startup (see `_make_server_entry` at line 308). + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00011.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00011.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00011.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00012.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00012.txt new file mode 100644 index 00000000..10fbd963 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00012.txt @@ -0,0 +1,5 @@ +The module **MUST NOT raise, ever** — it swallows every error by design so that the recording channel cannot break indexing. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00013.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00013.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00013.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00014.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00014.txt new file mode 100644 index 00000000..9ba23a8e --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00014.txt @@ -0,0 +1,5 @@ +I don't know. + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00015.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00015.txt new file mode 100644 index 00000000..832ba9ed --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/cand_c00015.txt @@ -0,0 +1,5 @@ +The fix uses a short-lived **lease file** that an indexing process refreshes on every progress tick. The idle watchdog is required to check this lease and must **NOT kill the embedder while any fresh lease exists** (its own reindex or another process's). + + +> build · longcat-2.0 + diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00000.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00000.txt new file mode 100644 index 00000000..f7384c39 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00000.txt @@ -0,0 +1,123 @@ +### src/core/error_envelope.py +"""Typed, actionable error envelope for MCP tools. + +Every tool failure should carry a stable ``code``, a human ``message``, and a +``next_action`` hint so an agent can branch and recover without guessing +(research 2026-09-25: Orisu/Perplexity/xAI MCP envelopes, GitHub error style +guide, LiveMCP-101 failure taxonomy). + +Shape:: + + {"ok": false, "error": {"code": "...", "message": "...", + "next_action": "...", "details": {...}}} + +Dependency-free (stdlib only) so any layer can import it without cycles. +""" +from __future__ import annotations + +import json +from dataclasses import asdict, dataclass, field +from enum import Enum +from typing import Any, Optional + +__all__ = ["ErrorCode", "ToolError", "tool_error", "error_json", "playbook"] + + +class ErrorCode(str, Enum): + VALIDATION = "VALIDATION_FAILED" + NOT_FOUND = "NOT_FOUND" + AUTH = "AUTH_REQUIRED" + TIMEOUT = "TIMEOUT" + RATE_LIMIT = "RATE_LIMIT" + INDEX_NOT_READY = "INDEX_NOT_READY" + REINDEX_IN_PROGRESS = "REINDEX_IN_PROGRESS" + DEPENDENCY_DOWN = "DEPENDENCY_DOWN" + CONFLICT = "CONFLICT" + INTERNAL = "INTERNAL_ERROR" + + +# The "what to do about it" for each code. Agents branch on `code`; humans read +# `next_action`. Keep each hint concrete and single-step. +_PLAYBOOK: dict[ErrorCode, str] = { + ErrorCode.VALIDATION: "Fix the arguments and retry.", + ErrorCode.NOT_FOUND: ( + "Verify the identifier. If the file is new/unchanged index is stale, " + "call intel_trigger_reindex(mode='incremental') and retry after completion." + ), + ErrorCode.AUTH: ( + "Set/refresh the credential in .env and restart the MCP. Do NOT blind-retry." + ), + ErrorCode.TIMEOUT: ( + "Bounded operation exceeded its limit. Retry once; if it repeats, reduce " + "scope (fewer files / smaller limit) or raise the relevant *_TIMEOUT env." + ), + ErrorCode.RATE_LIMIT: "Back off; the server already retries with backoff.", + ErrorCode.INDEX_NOT_READY: ( + "Index is empty or building. Call intel_trigger_reindex(mode='incremental') " + "and poll intel_get_job_status until 'completed'." + ), + ErrorCode.REINDEX_IN_PROGRESS: ( + "A reindex is running; searches fast-fail by design. Poll " + "intel_get_job_status(job_id) and retry when it completes." + ), + ErrorCode.DEPENDENCY_DOWN: ( + "Embedder/reranker unavailable. The server auto-revives it; retry in a " + "few seconds (first revive can take ~18s)." + ), + ErrorCode.CONFLICT: "Re-read the latest state, reconcile, and retry.", + ErrorCode.INTERNAL: ( + "Retry once. If it persists, call get_logs and report details.request_id." + ), +} + + +@dataclass +class ToolError: + code: str + message: str + next_action: str + details: dict = field(default_factory=dict) + + def to_dict(self) -> dict: + return {"ok": False, "error": asdict(self)} + + def to_json(self) -> str: + return json.dumps(self.to_dict(), ensure_ascii=False) + + +def playbook(code: ErrorCode | str) -> str: + """Return the canonical next_action for a code (raises on unknown).""" + if not isinstance(code, ErrorCode): + code = ErrorCode(code) + return _PLAYBOOK[code] + + +def tool_error( + code: ErrorCode | str, + message: str, + *, + detail: str = "", + details: Optional[dict] = None, + next_action: Optional[str] = None, +) -> ToolError: + """Build a typed, actionable error. + + Args: + code: one of ErrorCode (or its value). + message: human-readable cause. + detail: extra context appended to the playbook hint. + details: machine-readable extras (tool, missing, present, ...). + next_action: override the playbook hint (use sparingly). + """ + if not isinstance(code, ErrorCode): + code = ErrorCode(code) # raises ValueError on unknown code + hint = next_action if next_action is not None else _PLAYBOOK[code] + if detail: + hint = f"{hint} ({detail})" + return ToolError(code=code.value, message=message, next_action=hint, + details=details or {}) + + +def error_json(code: ErrorCode | str, message: str, **kwargs: Any) -> str: + """Convenience: ``tool_error(...).to_json()``.""" + return tool_error(code, message, **kwargs).to_json() diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00001.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00001.txt new file mode 100644 index 00000000..91a8be45 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00001.txt @@ -0,0 +1,139 @@ +### experiments/misc_probes/exp_timeout_cancel_mechanism.py +"""Experiment: does an in-thread timeout actually bound a non-cancellable call? + +Reproduces the mechanism behind the IVF finalize hang: + - `Future.result(timeout=)` does NOT stop a running thread; + - `ThreadPoolExecutor.shutdown(wait=True)` (in a finally) JOINS that thread; + - hence a "timeout guard" that ends in wait=True cannot fire. + +HYPOTHESIS (H1): with a worker that sleeps longer than the timeout, + (A) result(timeout=T) raises after ~T while the worker KEEPS running; + (B) shutdown(wait=True) then blocks for the REMAINING worker time (the hang); + (C) shutdown(wait=False) returns immediately. +HYPOTHESIS (H2): real bounding requires a daemon thread (abandon) or a + separate process with hard kill. + +Run: python experiments/misc_probes/exp_timeout_cancel_mechanism.py +""" +from __future__ import annotations + +import sys +import threading +import time + +if sys.stdout.encoding != "utf-8": + try: + sys.stdout.reconfigure(encoding="utf-8") + except Exception: + pass + +WORK = 6.0 # seconds the "native call" runs +TMO = 1.0 # the timeout we pretend to enforce + + +def _busy_box(tag: str, seconds: float = WORK) -> None: + time.sleep(seconds) + + +def part_a_result_timeout_does_not_cancel() -> tuple[float, bool, float]: + from concurrent.futures import ThreadPoolExecutor + + ex = ThreadPoolExecutor(max_workers=1) + t0 = time.perf_counter() + fut = ex.submit(_busy_box, "A") + timed_out = False + try: + fut.result(timeout=TMO) + except Exception: + timed_out = True + t_after_result = time.perf_counter() - t0 + still_running = any(th.is_alive() for th in threading.enumerate() + if th is not threading.current_thread()) + # emulate the code's finally: + t1 = time.perf_counter() + ex.shutdown(wait=True) + t_shutdown_wait_true = time.perf_counter() - t1 + return t_after_result, still_running, t_shutdown_wait_true + + +def part_c_shutdown_no_wait() -> float: + from concurrent.futures import ThreadPoolExecutor + + ex = ThreadPoolExecutor(max_workers=1) + ex.submit(_busy_box, "C", 20.0) # deliberately longer than the test + time.sleep(0.2) + t1 = time.perf_counter() + ex.shutdown(wait=False) + return time.perf_counter() - t1 + + +def part_d_daemon_thread_abandon() -> tuple[float, bool]: + t = threading.Thread(target=_busy_box, args=("D", 20.0), daemon=True) + t.start() + t0 = time.perf_counter() + t.join(timeout=TMO) + return time.perf_counter() - t0, t.is_alive() + + +def _child_sleep() -> None: # pragma: no cover - subprocess entry + time.sleep(60) + + +def part_e_process_hard_kill() -> float: + import multiprocessing as mp + + ctx = mp.get_context("spawn") + p = ctx.Process(target=_child_sleep) + p.start() + t0 = time.perf_counter() + p.join(timeout=TMO) + killed = False + if p.is_alive(): + p.terminate() + p.join(timeout=5) + killed = True + dt = time.perf_counter() - t0 + assert killed, "process should have been killed" + return dt + + +def main() -> int: + print(f"config: work={WORK}s timeout={TMO}s\n") + + a_t, a_alive, a_shutdown = part_a_result_timeout_does_not_cancel() + print("A) result(timeout) then finally shutdown(wait=True) [CURRENT CODE]") + print(f" raised timeout after : {a_t:.2f}s (expected ~{TMO}s)") + print(f" worker still running : {a_alive}") + print(f" shutdown(wait=True) : blocked {a_shutdown:.2f}s (HANG)") + print(f" => total before return ~ {a_t + a_shutdown:.2f}s " + f"(should have been ~{TMO}s)\n") + + c_t = part_c_shutdown_no_wait() + print("C) shutdown(wait=False)") + print(f" returned in : {c_t:.3f}s (does not join)\n") + + d_t, d_alive = part_d_daemon_thread_abandon() + print("D) daemon thread + join(timeout)") + print(f" returned in : {d_t:.2f}s; worker alive={d_alive} " + f"(abandoned; daemon => process can exit)\n") + + e_t = part_e_process_hard_kill() + print("E) separate process + terminate() on timeout") + print(f" returned in : {e_t:.2f}s; child killed=True\n") + + verdict_h1 = (a_t < 2.0) and a_alive and (a_shutdown > 2.0) + verdict_h2 = (d_t < 2.0) and (e_t < 2.5) + print(f"H1 (timeout cannot cancel; wait=True hangs): " + f"{'CONFIRMED' if verdict_h1 else 'REFUTED'}") + print(f"H2 (daemon/process bound the call): " + f"{'CONFIRMED' if verdict_h2 else 'REFUTED'}") + return 0 if (verdict_h1 and verdict_h2) else 1 + + +if __name__ == "__main__": + try: + sys.exit(main()) + except Exception: + import traceback + traceback.print_exc() + sys.exit(1) diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00002.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00002.txt new file mode 100644 index 00000000..563b1161 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00002.txt @@ -0,0 +1,323 @@ +### src/core/quiet_break_gate.py +"""Quiet-break gate — детерминированный детектор «тихого слома» графа. + +На коммите (через `graph_query(action="isolation")` + opencode-хук) отвечает на вопрос: +вносит ли ПРАВКА узел графа в изоляцию (осиротение), о котором молчит обычный ревью? + +Два ДЕЛЬТА-сигнала (абсолютная изоляция не годится: 1535/4081 функций репо имеют +0 входящих CALLS — декораторы, entry points, callbacks; это шум, не сигнал): + +A. removed_last_caller — символ, ВСЁ ЕЩЁ определённый в дереве, теряет последних + вызывающих: правка удаляет вызовы `name(`, а среди оставшихся вызывающих (по + persisted-графу) ни один больше не зовёт `name` в рабочем дереве. +B. new_orphan — НОВАЯ функция/метод (из diff), которую никто не вызывает в изменённых + файлах и которая не является entry point / декоратором-регистрантом. + +Всё ЧТЕНИЕ (git diff + persisted PropertyGraph), индекс НЕ мутируется. При отсутствии +git/графа/изменений — status "unavailable"/"empty", findings пусты, гейт НЕ срабатывает +(fail-open, причина логируется): инфраструктурный сбой не должен блокировать коммит. + +Границы (см. Red Team в .agent_task_state): rename-guard (def удалён в том же diff), +whitelist (main/run/handle/test_*/step_*/setup/teardown/dunder), decorated-skip, +comment-строки игнорируются, same-name — консервативно. +""" +from __future__ import annotations + +import logging +import re +import subprocess +from pathlib import Path +from typing import Any, Dict, List, Optional, Set, Tuple + +from src.core.graph import EdgeType, NodeLabel, PropertyGraph + +logger = logging.getLogger("mscodebase_server.quiet_break") + +__all__ = ["run_quiet_break_gate", "parse_diff", "is_whitelisted"] + +# entry points и регистранты — не считаются «сиротами» (зовутся фреймворком/декоратором) +_ENTRY_POINT_NAMES = { + "main", "run", "start", "handle", "entry", "bootstrap", "init_app", + "setup", "teardown", "setup_module", "teardown_module", "setup_method", + "teardown_method", "setup_class", "teardown_class", +} +_WHITELIST_PREFIXES = ("test_", "step_") + +_PY_KEYWORDS = { + "if", "elif", "else", "for", "while", "return", "yield", "with", "as", + "in", "is", "not", "and", "or", "lambda", "def", "class", "import", + "from", "raise", "assert", "del", "global", "nonlocal", "await", "async", + "match", "case", "print", "super", "type", "range", "len", "str", "int", + "float", "list", "dict", "set", "tuple", "bool", "open", "getattr", + "setattr", "isinstance", "enumerate", "zip", "sorted", "any", "all", +} + +_DEF_RE = re.compile(r"^\+\s*(?:async\s+)?def\s+([A-Za-z_]\w*)\s*\(") +_REMOVED_DEF_RE = re.compile(r"^-\s*(?:async\s+)?def\s+([A-Za-z_]\w*)\s*\(") +_CALL_RE = re.compile(r"\b([A-Za-z_]\w*)\s*\(") + +_GRAPH_LABELS = (NodeLabel.FUNCTION, NodeLabel.METHOD) + + +def is_whitelisted(name: str, decorated: bool = False) -> bool: + """True — имя является entry point / тест-хелпером / регистрантом (не «сирота»).""" + if decorated: + return True + if name in _ENTRY_POINT_NAMES: + return True + if any(name.startswith(p) for p in _WHITELIST_PREFIXES): + return True + return name.startswith("__") and name.endswith("__") + + +def parse_diff(diff_text: str) -> Dict[str, Any]: + """Разбирает unified-diff на дельта-сигналы. + + Returns dict: + removed_calls: {name: [file, ...]} + removed_defs: {name, ...} (def-строки, УДАЛЁННЫЕ этим diff — guard A) + new_defs: [{name, file, decorated}, ...] + changed_files: [rel-posix, ...] + """ + removed_calls: Dict[str, List[str]] = {} + removed_defs: Set[str] = set() + new_defs: List[Dict[str, Any]] = [] + changed_files: List[str] = [] + seen_files: Set[str] = set() + + cur_file = "" + prev_added = "" + + for raw in diff_text.splitlines(): + if raw.startswith("+++ "): + target = raw[4:].strip() + if target.startswith("b/"): + target = target[2:] + if target and target != "/dev/null": + cur_file = target + if cur_file not in seen_files: + seen_files.add(cur_file) + changed_files.append(cur_file) + prev_added = "" + continue + if raw.startswith("--- ") or raw.startswith("diff --git") or raw.startswith("@@"): + continue + + if raw.startswith("+"): + stripped = raw[1:].strip() + m = _DEF_RE.match(raw) + if m: + new_defs.append({ + "name": m.group(1), + "file": cur_file, + "decorated": prev_added.startswith("@"), + }) + prev_added = stripped + elif raw.startswith("-"): + if raw.startswith("---"): + continue + stripped = raw[1:].strip() + if stripped.startswith("#"): + prev_added = "" + continue + md = _REMOVED_DEF_RE.match(raw) + if md: + removed_defs.add(md.group(1)) + for name in _CALL_RE.findall(stripped): + if name in _PY_KEYWORDS: + continue + removed_calls.setdefault(name, []).append(cur_file) + prev_added = "" + else: + # контекстная строка + if raw.startswith(" "): + prev_added = raw[1:].strip() + + return { + "removed_calls": removed_calls, + "removed_defs": removed_defs, + "new_defs": new_defs, + "changed_files": changed_files, + } + + +def _rel_posix(path: str, root: Path) -> str: + if not path: + return "" + try: + return Path(path).resolve().relative_to(root).as_posix() + except (ValueError, OSError): + pass + p = Path(path) + for parent in p.parents: + try: + return p.relative_to(parent).as_posix() + except ValueError: + continue + return p.as_posix() + + +def _incoming_callers(pg: PropertyGraph, qname: str) -> List[Any]: + callers: List[Any] = [] + for et in (EdgeType.CALLS, EdgeType.ASYNC_CALLS): + try: + for node, _edge, _depth in pg.get_neighbors( + qname, edge_type=et, direction="incoming", max_depth=1 + ): + callers.append(node) + except Exception: # noqa: BLE001 — узел без рёбер/битый граф: не повод падать + continue + return callers + + +def _called_in_file(name: str, content: str) -> bool: + call_re = re.compile(r"\b" + re.escape(name) + r"\s*\(") + def_re = re.compile(r"\bdef\s+" + re.escape(name) + r"\s*\(") + for m in call_re.finditer(content): + line_start = content.rfind("\n", 0, m.start()) + 1 + if def_re.search(content[line_start:m.end()]): + continue # это сама def-строка, не вызов + return True + return False + + +def _git_diff(root: Path) -> Tuple[Optional[str], str]: + """git diff: staged если непусто, иначе HEAD. + + Returns (diff|None, base): None — не git/ошибка (fail-open); "" — git есть, + но изменений нет (status empty). + """ + flags = getattr(subprocess, "CREATE_NO_WINDOW", 0) + is_repo = False + for base_args, base_name in ((("--cached",), "cached"), (("HEAD",), "HEAD")): + try: + proc = subprocess.run( + ["git", "diff", "--no-color", "--unified=3", *base_args], + cwd=str(root), + capture_output=True, + timeout=20, + creationflags=flags, + ) + except Exception as exc: # noqa: BLE001 — git может отсутствовать/зависнуть + logger.warning("quiet_break: git diff failed (%s): %s", base_name, exc) + return None, "" + if proc.returncode == 0: + is_repo = True + out = proc.stdout.decode("utf-8", "replace") + if out.strip(): + return out, base_name + if is_repo: + return "", "none" + return None, "" + + +def _empty(status: str, message: str = "") -> Dict[str, Any]: + return { + "status": status, + "message": message, + "base": "", + "counts": {"removed_last_caller": 0, "new_orphan": 0}, + "findings": [], + "changed_files": [], + } + + +def run_quiet_break_gate( + project_root: Path, + pg: Optional[PropertyGraph], + diff_text: Optional[str] = None, +) -> Dict[str, Any]: + """Главная точка. Возвращает dict с findings; никогда не бросает наружу.""" + root = Path(project_root).resolve() + if pg is None: + return _empty("unavailable", "PropertyGraph not available") + db_path = getattr(pg, "_db_path", None) or getattr(pg, "path", None) + if not db_path or not Path(str(db_path)).exists(): + return _empty("unavailable", "graph db not found") + + if diff_text is None: + diff_text, base = _git_diff(root) + if diff_text is None: + return _empty("unavailable", "git not available") + else: + base = "provided" + if not diff_text.strip(): + return _empty("empty", "no changes") + + parsed = parse_diff(diff_text) + changed = set(parsed["changed_files"]) + removed_defs = parsed["removed_defs"] + findings: List[Dict[str, Any]] = [] + + # ── A: removed_last_caller ── + for name, locs in parsed["removed_calls"].items(): + if name in removed_defs or name in _PY_KEYWORDS: + continue + try: + nodes = [n for n in pg.find_nodes(name_pattern=name, limit=50) + if n.label in _GRAPH_LABELS] + except Exception as exc: # noqa: BLE001 — граф недоступен по узлу + logger.debug("quiet_break: find_nodes(%s) failed: %s", name, exc) + continue + for node in nodes: + callers = _incoming_callers(pg, node.qualified_name) + if not callers: + continue # уже был без вызывающих — не наш дельта-сигнал + alive = False + for c in callers: + cf = getattr(c, "file_path", "") or "" + if not cf: + alive = True + break + rel = _rel_posix(cf, root) + if rel not in changed: + alive = True + break + fp = root / rel + try: + content = fp.read_text(encoding="utf-8", errors="replace") + except OSError: + alive = True # не смогли прочитать — консервативно считаем живым + break + if _called_in_file(name, content): + alive = True + break + if not alive: + findings.append({ + "kind": "removed_last_caller", + "symbol": node.name, + "file": node.file_path, + "removed_call_lines": locs, + }) + + # ── B: new_orphan ── + contents: Dict[str, str] = {} + for rel in changed: + fp = root / rel + try: + contents[rel] = fp.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + for d in parsed["new_defs"]: + if is_whitelisted(d["name"], d["decorated"]): + continue + in_changed = any(_called_in_file(d["name"], txt) for txt in contents.values()) + if not in_changed: + findings.append({ + "kind": "new_orphan", + "symbol": d["name"], + "file": d["file"], + "decorated": d["decorated"], + }) + + counts = { + "removed_last_caller": sum(1 for f in findings if f["kind"] == "removed_last_caller"), + "new_orphan": sum(1 for f in findings if f["kind"] == "new_orphan"), + } + return { + "status": "ok", + "base": base, + "counts": counts, + "findings": findings, + "changed_files": parsed["changed_files"], + } diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00003.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00003.txt new file mode 100644 index 00000000..3b6e141c --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00003.txt @@ -0,0 +1,209 @@ +### src/providers/reranker/reranker_scoring.py +"""Scoring helpers для MultiProviderReranker. + +Выделены из multi_provider.py для уменьшения god-object. +""" +from __future__ import annotations + +import json +import logging +import math +import re +from typing import Any, Dict, List + +logger = logging.getLogger(__name__) + +# Регулярка для извлечения JSON-массива scores из ответа +_SCORES_JSON_RE = re.compile(r'\{\s*"scores"\s*:\s*\[.*?\]\s*\}', re.DOTALL) +# Извлечение отдельных объектов {"index": N, "score": F} +_SCORE_ITEM_RE = re.compile( + r'\{\s*"index"\s*:\s*(\d+)\s*,\s*"score"\s*:\s*([+-]?\d*\.?\d+(?:[eE][+-]?\d+)?)\s*\}' +) + + +def cosine_similarity(vec_a: List[float], vec_b: List[float]) -> float: + """Вычисляет cosine similarity между двумя векторами.""" + if not vec_a or not vec_b or len(vec_a) != len(vec_b): + return 0.0 + + dot = sum(a * b for a, b in zip(vec_a, vec_b)) + norm_a = sum(a * a for a in vec_a) ** 0.5 + norm_b = sum(b * b for b in vec_b) ** 0.5 + + if norm_a == 0 or norm_b == 0: + return 0.0 + + return dot / (norm_a * norm_b) + + +def validate_scores(scores: List[Any]) -> List[Dict[str, Any]]: + """Валидирует и нормализует список скоров. + + Контракт (ground truth для мутационных тестов, evalmut-инвариант): + - index: неотрицательное ЦЕЛОЕ. bool отбрасывается (bool — подкласс int); + float принимается только при целочисленном значении — иначе int() + молча переставил бы скор (напр. 2.7 -> 2); негативный индекс не + ссылается ни на один чанк. + - score: конечное число. NaN/Inf отбрасываются — это НЕ «больше 1», + а отсутствие оценки: min/max-clamp молча превращал NaN в 1.0 + (максимальный скор неоценённому чанку). Вне [0,1] — clamp + (by design: логиты реранкера нормализуются). + Элементы, не прошедшие контракт, отбрасываются (fail-safe: потерянный + скор лучше переставленного). + """ + validated = [] + for item in scores: + if not isinstance(item, dict): + continue + idx = item.get("index") + score = item.get("score") + # bool — подкласс int: True не является валидным индексом/скором + if isinstance(idx, bool) or isinstance(score, bool): + continue + if not isinstance(idx, (int, float)) or not isinstance(score, (int, float)): + continue + if isinstance(idx, float) and not idx.is_integer(): + continue # 2.7 -> int() молча переставил бы скор на соседний чанк + index = int(idx) + if index < 0: + continue + if not math.isfinite(score): + continue # NaN/Inf проходили isinstance и становились 1.0/0.0 + validated.append({"index": index, "score": max(0.0, min(1.0, float(score)))}) + return validated + + +def _finalize_scores( + parsed: List[Dict[str, Any]], + raw: str, + *, + single_decline: bool, +) -> List[Dict[str, Any]]: + """Общий выходной фильтр путей парсера: decline при недостоверном извлечении. + + Дубликаты индексов — сигнал «пример формата + реальные скоры» (пример в + промпте всегда {"index": 0, ...} и совпадёт с реальным 0), а также битого + ответа LLM — decline во всех путях. Единственный объект — decline ТОЛЬКО + на regex-пути без обёртки (single_decline=True), где объект мог быть + примером формата из объяснения; внутри полной обёртки {"scores": [...]} + одиночный объект легитимен (реальный ответ для одного чанка). + Возвращаем [] (fail-safe: не сортировать вообще, чем по мусору). + """ + if not parsed: + return [] + indices = [s["index"] for s in parsed] + if len(indices) != len(set(indices)): + logger.warning( + f"⚠️ Дублирующиеся индексы скоров (вероятный пример формата " + f"в ответе), decline: {raw[:200]}..." + ) + return [] + if single_decline and len(parsed) == 1: + logger.warning( + f"⚠️ Единичный объект score без обёртки scores — вероятный " + f"пример формата, decline: {raw[:200]}..." + ) + return [] + return parsed + + +def parse_scores_json(raw: str) -> List[Dict[str, Any]]: + """Парсит JSON со скорами из ответа LLM. + + Поддерживает: + 1. Чистый JSON: {"scores": [{"index": 0, "score": 0.95}, ...]} + 2. JSON в markdown-блоке: ```json\n{...}\n``` + 3. JSON с окружающим текстом (поиск через regex) + + Returns: + Список dict'ов [{"index": int, "score": float}, ...] + """ + if not raw: + return [] + + # Попытка 1: прямой JSON-парсинг + try: + data = json.loads(raw) + scores = data.get("scores", []) + if isinstance(scores, list) and scores: + return _finalize_scores(validate_scores(scores), raw, single_decline=False) + except (json.JSONDecodeError, TypeError): + pass + + # Попытка 2: извлечение из markdown-блока + md_match = re.search(r"```(?:json)?\s*\n?(.*?)\n?```", raw, re.DOTALL) + if md_match: + try: + data = json.loads(md_match.group(1)) + scores = data.get("scores", []) + if isinstance(scores, list) and scores: + return _finalize_scores(validate_scores(scores), raw, single_decline=False) + except (json.JSONDecodeError, TypeError): + pass + + # Попытка 3: поиск JSON-объекта через regex + json_match = _SCORES_JSON_RE.search(raw) + if json_match: + try: + data = json.loads(json_match.group(0)) + scores = data.get("scores", []) + if isinstance(scores, list) and scores: + return _finalize_scores(validate_scores(scores), raw, single_decline=False) + except (json.JSONDecodeError, TypeError): + pass + + # Попытка 4: извлечение отдельных объектов score (fallback: контракт + # промпта требует полный {"scores": [...]}, но отдельные LLM отвечают + # голыми объектами). Тот же контракт, что в путях 1-3 (clamp+фильтры). + items = _SCORE_ITEM_RE.findall(raw) + if items: + return _finalize_scores( + validate_scores( + [{"index": int(idx), "score": float(score)} for idx, score in items] + ), + raw, + single_decline=True, + ) + + logger.warning( + f"Не удалось извлечь scores из ответа реранкера: {raw[:200]}..." + ) + return [] + + +def apply_scores( + chunks: List[Dict[str, Any]], + scores: List[Dict[str, Any]], + top_n: int, +) -> List[Dict[str, Any]]: + """Применяет скоры реранкера к чанкам и сортирует.""" + n = len(chunks) + # Диагностика «осиротевших» индексов (>= n): парсер не знает число чанков, + # поэтому скор, не ссылающийся ни на один чанк, молча терялся. Не влияет + # на сортировку остальных — но обязан быть видимым в логе (evalmut: + # coverage gap, а не тихая потеря). + orphaned = [s["index"] for s in scores if s.get("index", 0) >= n] + if orphaned: + logger.warning( + f"⚠️ Скоры с индексами вне диапазона чанков [0,{n}): {orphaned} — " + f"вероятно, LLM вернул индексы несуществующих чанков." + ) + score_map = {s["index"]: s["score"] for s in scores} + for i, chunk in enumerate(chunks): + chunk["reranker_score"] = score_map.get(i, 0.0) + + sorted_chunks = sorted( + chunks, + key=lambda c: c.get("reranker_score", 0.0), + reverse=True, + ) + + return sorted_chunks[:top_n] + + +__all__ = [ + "cosine_similarity", + "validate_scores", + "parse_scores_json", + "apply_scores", +] diff --git a/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00004.txt b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00004.txt new file mode 100644 index 00000000..ef34f3b2 --- /dev/null +++ b/experiments/4A_unit_of_return/results/f5relang/en/work/ctx_c00004.txt @@ -0,0 +1,946 @@ +### src/core/intelligence/verify_on_read.py +""" +ADR-0003 Verify-On-Read — Lazy Validation Layer для Project Memory. + +Ленивая проверка узлов при ИЗВЛЕЧЕНИИ (retrieval), до формирования системного +промпта. Вектор проверки смещён с момента отклика/записи на момент чтения: +`intel_get_project_memory` (хук в layer.py) пропускает узлы через этот слой. + +Правила (ADR-0003, решения владельца 2026-08-11): +- Узлы с checkable-якорями (`file:`/`import:`/`env:`/`pkg:`) сверяются с кодовой базой: + * все якоря найдены -> VERIFIED (persist); + * прямое отрицательное тестирование якоря (указан, но не существует) + -> REFUTED с причиной SILENT_ABSENCE_ON_READ (persist, retract_source); + * якорей нет (внешнее окружение/предпочтения) -> INCONCLUSIVE, статус + сохраняется (ACTIVE/VERIFIED) — предохранитель от ложных отзывов + истинных фактов, не оставляющих следов в рабочей директории. +- Кэш вердиктов: ключ = hash(node_id + commit_sha) (git rev-parse HEAD; + fallback — максимальный mtime src-дерева). Неизменившийся репозиторий + не перепроверяется (cache hit ~0ms); смена HEAD естественно инвалидирует + ТОЛЬКО свои записи (per-node keying, без TTL) -> повторная проверка узла + при следующем чтении (исключает stale-VERIFIED). +- Guard проза-«import X» (ADR-0005, 2026-08-14): частотное англ. слово без + src-импорта (напр. «import path») якорем не становится — ни на write-path + (не сохраняется), ни на read-path (не отзывает узел). Редкие слова + (grafana/celery) сохраняются — SILENT-детекция и smoke-негатив живы. + Явные data.anchors (намеренные) не фильтруются. +- Бюджет латентности: <= budget_ms на весь проход; при превышении + необработанные узлы остаются как есть (INCONCLUSIVE-семантика) и передаются + в контекст без отзыва (graceful degradation). +- Счётчики MATCHED/DELIVERED (Том, 2026-08-16): per-node накопительные + счётчики в verify_cache.json (ключ node_id, переживают HEAD). matched — + сколько раз узел был виден в проходе, delivered — сколько раз вердикт + реально разрешён (свежая проверка или cache-hit). matched>=2 && delivered==0 + -> starved (систематическое голодание по бюджету), в отличие от бага + якорей/парсера. Ресипт несёт starved_nodes. +- Отпечаток кодовой базы (импорты/файлы/env-ключи) строится один раз на HEAD + и переиспользуется между процессами через verify_cache.json. Первое чтение + после смены кода платит rebuild (~<=500ms); steady-state — cache hit ~0ms. + +Аудит: авто-отзывы пишутся тем же путём, что ручные (retract_reason, +retracted_at) + маркер retract_source="verify_on_read" и имя проваленного +якоря в причине. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +import os +import re +import subprocess +import time +from datetime import datetime, timedelta +from pathlib import Path +from typing import Any, Callable, Dict, List, Optional, Set, Tuple, Union + +if os.name == "nt": + import subprocess as _sp + + _CREATE_NO_WINDOW = getattr(_sp, "CREATE_NO_WINDOW", 0) +else: + _CREATE_NO_WINDOW = 0 + +logger = logging.getLogger("MSCodeBase.Intelligence.VerifyOnRead") + +REASON_SILENT_ABSENCE = "SILENT_ABSENCE_ON_READ" +RETRACT_SOURCE = "verify_on_read" + +VERDICT_FOUND = "FOUND" +VERDICT_NOT_FOUND = "NOT_FOUND" +VERDICT_INCONCLUSIVE = "INCONCLUSIVE" + +STATUS_ACTIVE = "ACTIVE" +STATUS_VERIFIED = "VERIFIED" +STATUS_REFUTED = "REFUTED" +STATUS_SUPERSEDED = "SUPERSEDED" + +DEFAULT_BUDGET_MS = 50.0 +CACHE_FILENAME = "verify_cache.json" +HEAD_TTL_SEC = 30.0 # пере-резолв HEAD не чаще раза в 30с на инстанс (git ~50-100ms) + +# H3 TTL-гниение (doc 10-continuous-verification): узел, НЕ проверенный в проходе, +# чей след проверки (verified_at/last_checked) старше TTL_STALE_DAYS → метка +# stale_ttl «не подтверждён за N дней». N из распределения verified_at live-памяти +# (медиана 22, потолок 31 день) — мерено 2026-09-11, плюс коммитовый ритм ежедневный. +TTL_STALE_DAYS = int(os.getenv("VOR_TTL_DAYS", "30")) +# Физическую запись last_checked rate-limit'им (H1 idle гоняет каждый тик; без +# порога каждый тик переписывал бы project_memory.json). Порог — возраст следа. +LAST_CHECKED_MIN_INTERVAL = float(os.getenv("VOR_LAST_CHECKED_INTERVAL_SEC", str(6 * 3600))) +_TS_FMT = "%Y-%m-%d %H:%M:%S" + +MANIFEST_CACHE_TTL_S = 5.0 # write-path зовёт extract_anchors per-node; полный скан манифестов (B-1) дороже 3 файлов старого парсера + +# ── Лёгкие regex-якоря (без LLM) ── +# Для сканирования исходников: импорты на строке (реальный код). +_IMPORT_RE = re.compile(r"^\s*(?:import|from)\s+([a-zA-Z_][a-zA-Z0-9_]*)", re.MULTILINE) +# Для текста claim: "import X" в любой позиции, "from X import Y" целиком. +_TEXT_IMPORT_RE = re.compile(r"\bimport\s+([a-zA-Z_][a-zA-Z0-9_.]*)") +_TEXT_FROM_RE = re.compile(r"\bfrom\s+([a-zA-Z_][a-zA-Z0-9_.]*)\s+import\b") +_ENV_PREFIX_RE = re.compile(r"(?:env:|\$)([A-Z][A-Z0-9_]{2,})") +_FILE_PREFIX_RE = re.compile(r"file:([^\s`'\";,]+)") +_PATH_RE = re.compile(r"\b(?:src/)?([\w.-]+[/\\\\][\w./\\\\-]+)\.(?:py|toml|json|md|yaml|yml|ini|cfg|env|db)\b") +# ADR-0005: явный синтаксис pkg:name и прозa-слово-кандидат (write-path capture). +_PKG_PREFIX_RE = re.compile(r"\bpkg:([A-Za-z0-9_.-]+)") +_PKG_WORD_RE = re.compile(r"[A-Za-z][A-Za-z0-9_.-]*") + + +def _norm_pkg(name: str) -> str: + """PEP 503 нормализация имени пакета: lowercase, `-_.` -> `-`, срез extras.""" + name = name.split("[", 1)[0].strip() + return re.sub(r"[-_.]+", "-", name.lower()) + + +def _pkg_in_pool(pool: Set[str], name: str) -> bool: + """Членство pkg:-имени в пуле манифеста: точное (npm/go имена с точками + и слэшами: `lodash.get`, `gopkg.in/yaml.v3`) ИЛИ PEP 503-нормализованное + (legacy python-имена, `PyYAML` -> `pyyaml`). Только PEP 503 давал бы ложный + NOT_FOUND -> REFUTED после B-1 wiring (мульти-экосистемные имена).""" + n = (name or "").strip().lower() + return n in pool or _norm_pkg(n) in pool + + +_MANIFEST_CACHE: Dict[str, Tuple[float, Set[str]]] = {} + + +def _load_manifest_packages(root: Path) -> Set[str]: + """Множество зависимостей проекта (ADR-0005, closed world). + + Делегирует мульти-экосистемному парсеру src.sources.manifest (Backlog B-1): + python/npm/go/cargo/maven/nuget/composer/gem + lockfile'ы (uv/yarn/bun/pnpm/...). + Lazy-import: экстракторы не грузятся при импорте VOR. TTL-кэш: write-path + вызывает extract_anchors на узел; устаревший кэш = пропуск пакета -> + INCONCLUSIVE (fail-open), не ложный REFUTED. + """ + key = str(Path(root).resolve()) + now = time.monotonic() + hit = _MANIFEST_CACHE.get(key) + if hit is not None and now - hit[0] < MANIFEST_CACHE_TTL_S: + return hit[1] + try: + from src.sources.manifest import manifest_packages + + pool = manifest_packages(Path(key)) + except Exception as exc: # noqa: BLE001 - сломанный экстрактор не валит write-path + logger.debug("verify_on_read: manifest_packages failed for %s: %s", key, exc) + pool = set() + _MANIFEST_CACHE[key] = (now, pool) + return pool + + +# ADR-0005 guard (C-гибрид): частые англ. слова, которые в прозе могут стоять +# после «import» как фразы («import path» = «путь импорта»), а не как команда +# импорта. Реальные stdlib-имена (time/os/json) тоже здесь: якорь сохраняется, +# если имя реально импортировано в src (см. _add_import_anchor). +_COMMON_WORDS = frozenset({ + "path", "data", "time", "file", "name", "value", "state", "type", "list", + "order", "work", "part", "form", "line", "code", "key", "user", "group", + "page", "view", "model", "class", "source", "target", "main", "status", + "error", "result", "input", "output", "config", "build", "release", + "version", "update", "change", "create", "delete", "add", "remove", + "open", "close", "save", "load", "read", "write", "print", "send", + "receive", "test", "debug", "log", "access", "process", "thread", "task", + "job", "event", "loop", "queue", "cache", "stack", "heap", "index", + "query", "search", "request", "response", "session", "token", "secret", + "auth", "login", "password", "admin", "role", "field", "column", "row", + "table", "schema", "network", "protocol", "format", "content", "header", + "body", "message", "channel", "stream", "batch", "graph", "node", "edge", + "size", "level", "count", "total", "average", "sum", "scope", "system", + "service", "client", "server", "runtime", "step", "run", "start", "end", + "case", "point", "module", "package", "function", "global", "local", + "public", "private", "static", "final", "active", "ready", "os", "io", + "re", "sys", "json", "csv", "xml", "yaml", "toml", "url", "http", "api", + "db", "ui", "cli", "ml", "ai", "sql", "dns", "ip", "tcp", "tls", "rest", +}) + + +_FINGERPRINT_CACHE: Dict[str, "_Fingerprint"] = {} + + +def _fingerprint_for(root: Path) -> "_Fingerprint": + """Кэш отпечатка для write-path guard (auto_collect_adrs вызывает часто). + + Fail-open по стойкости: устаревший кэш даёт дроп лишнего якоря -> INCONCLUSIVE, + а не ложный REFUTED (бias поста: false negative дешевле false positive). + """ + key = str(Path(root).resolve()) + fp = _FINGERPRINT_CACHE.get(key) + if fp is None: + fp = _Fingerprint(root=root) + _FINGERPRINT_CACHE[key] = fp + return fp + + +# ── Freshness для symbol-anchors (issue #21/#22, commit B) ── +# A symbol anchor can only REFUTE honestly when the index it was verified +# against provably reflects the current codebase. These two helpers gate that: +# * evaluate_freshness(build_head, current_head, dirty): pure decision — is the +# substrate fresh enough that "not found" is real evidence of absence? +# * resolve_head_dirty(root): (current_HEAD, dirty) or None when unresolvable. + + +def evaluate_freshness( + build_head: Optional[str], + current_head: Optional[str], + dirty: bool, +) -> bool: + """True only when the symbol index is provably fresh enough to REFUTE. + + REFUTE (absent) requires: an index build_head that equals the live HEAD, + a resolvable current HEAD, and a clean working tree. Any other state + (mismatched HEAD, legacy index with no build_head, non-git repo + / unresolvable HEAD, or a dirty tree) means absence is UNVERIFIABLE -> + the resolver must return None (INCONCLUSIVE), never False (REFUTED). + """ + if build_head is None: + return False # legacy index / no freshness mark -> unknown + if current_head is None: + return False # non-git / git failure -> unresolvable + if dirty: + return False # uncommitted edits may redefine the symbol + return build_head == current_head + + +def resolve_head_dirty(root: Union[str, Path]) -> Optional[Tuple[str, bool]]: + """Current HEAD plus dirty-flag for a git repo; None when unresolvable. + + Returns (head, dirty): head is the bare 40-char rev-parse HEAD, dirty is + True when ``git status --porcelain`` is non-empty (uncommitted changes). + None when git is unavailable / not a repo / any error — caller must treat + that as INCONCLUSIVE. Both are resolved best-effort subprocess calls. + """ + root_p = Path(root).resolve() + if (root_p / ".git").exists() is False and not _is_git_worktree(root_p): + return None # no git metadata at all -> not resolvable + try: + proc = subprocess.Popen( + ["git", "rev-parse", "HEAD"], + cwd=str(root_p), + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + creationflags=_CREATE_NO_WINDOW, + ) + out, _ = proc.communicate(timeout=5) + if proc.returncode != 0: + return None + head = out.decode("utf-8", "replace").strip()[:40] + if not head: + return None + sproc = subprocess.Popen( + ["git", "status", "--porcelain"], + cwd=str(root_p), + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + creationflags=_CREATE_NO_WINDOW, + ) + sout, _ = sproc.communicate(timeout=5) + dirty = sproc.returncode == 0 and bool(sout.strip()) + return head, dirty + except Exception as exc: # noqa: BLE001 - любой сбой git -> не резолвится + logger.debug("resolve_head_dirty: git failed for %s: %s", root_p, exc) + return None + + +def _is_git_worktree(root: Path) -> bool: + """A subdir of a git worktree may have no own .git file; check nearest root.""" + try: + proc = subprocess.Popen( + ["git", "rev-parse", "--is-inside-work-tree"], + cwd=str(root), + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + creationflags=_CREATE_NO_WINDOW, + ) + out, _ = proc.communicate(timeout=5) + return proc.returncode == 0 and out.decode("utf-8", "replace").strip() == "true" + except Exception: # noqa: BLE001 + return False + + +class Anchor: + """Checkable-якорь узла: что именно проверяется в кодовой базе.""" + + __slots__ = ("kind", "value") + + def __init__(self, kind: str, value: str): + self.kind = kind # "file" | "import" | "env" | "pkg" + self.value = value + + def __repr__(self) -> str: # pragma: no cover - debug helper + return f"Anchor({self.kind}:{self.value})" + + def to_dict(self) -> Dict[str, str]: + return {"kind": self.kind, "value": self.value} + + @classmethod + def from_dict(cls, d: Dict[str, Any]) -> "Anchor": + return cls(str(d.get("kind", "")), str(d.get("value", ""))) + + +def extract_anchors( + node: Dict[str, Any], + project_root: Optional[Path] = None, + src_imports: Optional[Set[str]] = None, + read_path: bool = False, +) -> List[Anchor]: + """Извлекает checkable-якоря из data/claim узла (лёгкий regex, без LLM). + + Приоритет: явные `data.anchors` (пишутся при записи узла), затем синтаксис + в тексте claim/data: `file:path`, `import X` / `from X import y`, + `env:KEY` / `$KEY`, `pkg:name` (ADR-0005, закрытый мир манифеста), + пути с разделителями и расширением. + + Write-path (project_root задан): file-якоря фильтруются по существованию; + слова прозы, совпадающие с зависимостями манифеста, становятся pkg:-якорями + (fail-closed — слова вне манифеста якорем не становятся). + + project_root (write-path, P2-фикс): при передаче file-якоря, которых нет + относительно корня, отбрасываются — вольный текст коммитов даёт мусор + (слепленные пути «pyproject/extension.toml/__init__.py», относительные + «queries/__init__.py», завершающая пунктуация «__init__.py.»), а fail-closed + _classify превращает его в ЛОЖНЫЙ REFUTED. Read-path (None) — все якоря + классифицируются честно: удалённый файл = дрейф → REFUTED (фильтр не + отключает детекцию дрейфа). + """ + anchors: List[Anchor] = [] + seen: Set[Tuple[str, str]] = set() + + def _add(kind: str, value: str) -> None: + value = value.strip().strip("`'\"") + # Обрезка завершающей пунктуации: "src/.../__init__.py." -> ".../__init__.py" + value = value.rstrip(".,;:!?)]}") + if not value: + return + # P2: write-path хранит только существующие файлы (мусор из текста — мимо) + if kind == "file" and project_root is not None: + if not (project_root / value).is_file(): + return + key = (kind, value) + if key not in seen: + seen.add(key) + anchors.append(Anchor(kind, value)) + + # ADR-0005: write-path читает манифест для pkg:-capture (fail-closed). + pkg_pool: Optional[Set[str]] = None + if project_root is not None: + pkg_pool = _load_manifest_packages(project_root) + + # ADR-0005 guard (C-гибрид): src-импорты для отсева проза-«import X». + # Read-path (run()) передаёт fp.imports (свежий на HEAD); write-path — кэш. + src_pool = src_imports if src_imports is not None else ( + _fingerprint_for(project_root).imports if project_root is not None else None + ) + + def _add_import_anchor(value: str) -> None: + # Частотное англ. слово БЕЗ src-импорта = фраза («import path»), не якорь. + # Редкие слова (grafana/celery) сохраняются даже без src — SILENT-детекция + # и smoke-негативный контроль остаются рабочими. Явные data.anchors + # (намеренные) guard'ом не фильтруются. + if value in _COMMON_WORDS and src_pool is not None and value not in src_pool: + return + _add("import", value) + + data = node.get("data") or {} + if isinstance(data, dict): + raw_anchors = data.get("anchors") + if isinstance(raw_anchors, list): + for a in raw_anchors: + if isinstance(a, dict) and a.get("kind") in ("file", "import", "env", "pkg", "symbol"): + _add(str(a["kind"]), str(a.get("value", ""))) + # Fix (2026-09-11, 1-B): явные якоря = write-time capture, проза тела — + # история («X -> Y» в ADR-body). При read_path=True и наличии явных + # якорей проза НЕ сканируется: исторический путь не должен отзывать + # живой узел (ложный REFUTED после rename-sweep). + if read_path and anchors: + return anchors + parts: List[str] = [] + claim = data.get("claim") or "" + if claim: + parts.append(str(claim)) + for v in data.values(): + if isinstance(v, str) and v != claim: + parts.append(v) + text = "\n".join(parts) + elif isinstance(data, str): + text = data + else: + text = "" + + for m in _FILE_PREFIX_RE.finditer(text): + _add("file", m.group(1)) + for m in _PATH_RE.finditer(text): + value = m.group(0) + # Абсолютные/вложенные пути (drive-буква C:\, сегменты a\b\c, URL-схемы) + # — не проектные якоря: проверка идёт относительно корня проекта (ADR-0003). + if "://" in value or (m.start() > 0 and text[m.start() - 1] in (":", "\\", "/")): + continue + _add("file", value) + for m in _TEXT_IMPORT_RE.finditer(text): + _add_import_anchor(m.group(1)) + for m in _TEXT_FROM_RE.finditer(text): + _add_import_anchor(m.group(1)) + for m in _PKG_PREFIX_RE.finditer(text): + _add("pkg", m.group(1)) + for m in _ENV_PREFIX_RE.finditer(text): + _add("env", m.group(1)) + # Write-path (ADR-0005): слово прозы, чьё нормализованное имя есть в манифесте, + # становится pkg:-якорем. Fail-closed: слова НЕ в манифесте якорем не становятся + # (нет ложного REFUTED для исторических «мы перешли с X»). Read-path — только + # явный `pkg:` синтаксис (без present-trap). + if pkg_pool: + for m in _PKG_WORD_RE.finditer(text): + word = m.group(0) + if len(word) >= 2 and _pkg_in_pool(pkg_pool, word): + _add("pkg", word) + return anchors + + +class _Fingerprint: + """Отпечаток кодовой базы: root-импорты, файлы, env-ключи (один раз на HEAD).""" + + def __init__(self, root: Optional[Path] = None, data: Optional[Dict[str, Any]] = None): + if data is not None: + self.imports: Set[str] = set(data.get("imports", [])) + self.files: Set[str] = set(data.get("files", [])) + self.env_keys: Set[str] = set(data.get("env_keys", [])) + # ADR-0005: старый кэш без "packages" пересобирается (schema guard в + # _ensure_fingerprint) — пустой набор здесь привёл бы к ложным REFUTED. + self.packages: Set[str] = set(data.get("packages", [])) + self.build_ms: float = 0.0 + return + if root is None: + root = Path() + t0 = time.perf_counter() + imports: Set[str] = set() + files: Set[str] = set() + src = root / "src" + if src.is_dir(): + for p in src.rglob("*"): + if not p.is_file() or "__pycache__" in p.parts: + continue + rel = p.relative_to(root).as_posix() + files.add(rel) + if p.suffix == ".py": + try: + text = p.read_text(encoding="utf-8", errors="replace") + except OSError: + continue + for m in _IMPORT_RE.finditer(text): + imports.add(m.group(1).split(".")[0]) + env_keys: Set[str] = set() + if root is not None: + for name in (".env", ".env.example"): + ep = root / name + if ep.is_file(): + try: + for line in ep.read_text(encoding="utf-8", errors="replace").splitlines(): + line = line.strip() + if line and not line.startswith("#") and "=" in line: + env_keys.add(line.split("=", 1)[0].strip()) + except OSError: + continue + self.imports = imports + self.files = files + self.env_keys = env_keys + # ADR-0005: закрытый мир зависимостей (pyproject/requirements[-lock]). + self.packages = _load_manifest_packages(root) if root is not None else set() + self.build_ms = (time.perf_counter() - t0) * 1000.0 + + def to_dict(self) -> Dict[str, Any]: + return { + "imports": sorted(self.imports), + "files": sorted(self.files), + "env_keys": sorted(self.env_keys), + "packages": sorted(self.packages), + "build_ms": round(self.build_ms, 1), + } + + +class VerifyOnRead: + """Lazy Validation Layer (ADR-0003): фильтрация + переходы статусов при чтении. + + Один экземпляр на проект (см. `get_verifier`) — разделяет HEAD/отпечаток/кэш + между вызовами; переходы пишутся под тем же `write_lock`, что и ручные + операции памяти. + """ + + def __init__( + self, + project_root: Path, + store: Any, + write_lock: Any, + cache_file: Optional[Path] = None, + symbol_resolver: Optional[Callable[[str], Optional[bool]]] = None, + ): + self.root = project_root + self.store = store + self.lock = write_lock + # symbol_resolver(qname) -> True (exists) | False (gone) | None (unknown). + # Optional: intel layer injects a SymbolIndex-backed resolver; when None, + # a `symbol` anchor is UNVERIFIABLE -> classify INCONCLUSIVE (never VERIFIED). + self.symbol_resolver = symbol_resolver + if cache_file is None: + from src.core.artifact_paths import get_intelligence_dir + + cache_file = get_intelligence_dir(project_root) / CACHE_FILENAME + self.cache_file = cache_file + self._head: Optional[str] = None + self._fingerprint: Optional[_Fingerprint] = None + self._cache: Dict[str, Any] = { + "head": None, + "fingerprint": None, + "verdicts": {}, + "counters": {}, # per-node MATCHED/DELIVERED (Том) — ключ node_id, не head + } + self._head_cache: Dict[str, Any] = {"head": None, "ts": 0.0} + self._dirty_cache: Dict[str, Any] = {"dirty": False, "ts": 0.0} + self._load_cache() + + # ── HEAD и отпечаток ── + + def _resolve_head(self) -> str: + """HEAD с TTL-кэшем: git-подпроцесс не вызывается на каждом чтении. + + steady-state чтение (~0ms) — ключ не пересоздаётся; смена кода + детектится в пределах HEAD_TTL_SEC (пере-резолв), затем per-node + инвалидация по новому ключу. + """ + now = time.monotonic() + cached_head = self._head_cache.get("head") + if cached_head and now - self._head_cache.get("ts", 0.0) < HEAD_TTL_SEC: + return str(cached_head) + head = self._resolve_head_impl() + self._head_cache = {"head": head, "ts": now} + return head + + def _resolve_head_impl(self) -> str: + """git rev-parse HEAD (Popen+communicate, timeout); fallback — mtime src.""" + try: + proc = subprocess.Popen( + ["git", "rev-parse", "HEAD"], + cwd=str(self.root), + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + creationflags=_CREATE_NO_WINDOW, + ) + out, _ = proc.communicate(timeout=5) + if proc.returncode == 0: + return out.decode("utf-8", "replace").strip()[:40] + except Exception as exc: # noqa: BLE001 - любой сбой git — fallback + logger.debug("verify_on_read: git rev-parse failed: %s", exc) + max_m = 0 + src = self.root / "src" + if src.is_dir(): + for p in src.rglob("*"): + try: + max_m = max(max_m, p.stat().st_mtime_ns) + except OSError: + continue + return f"mtime:{max_m}" + + def _is_dirty(self) -> bool: + """Dirty-флаг рабочего дерева с TTL-кэшем (git subprocess не на каждое чтение). + + Dirty = незакоммиченные изменения (`git status --porcelain` непуст). + Non-git / нерезолвится -> False (fallback на mtime-head в _resolve_head + уже детектит правки файлов — dirty не нужен для инвалидации). + """ + now = time.monotonic() + cached = self._dirty_cache.get("dirty") + if now - self._dirty_cache.get("ts", 0.0) < HEAD_TTL_SEC: + return bool(cached) + res = resolve_head_dirty(self.root) + dirty = False if res is None else res[1] + self._dirty_cache = {"dirty": dirty, "ts": now} + return dirty + + def _ensure_fingerprint(self, head: str, dirty: bool = False) -> _Fingerprint: + # Dirty tree: незакоммиченные правки могут менять импорты/файлы — отпечаток + # строится заново КАЖДЫЙ проход (кэш по HEAD не действует: закоммиченный + # отпечаток лгал бы про живое дерево, а кэш отпечатка внутри dirty-интервала + # воспроизводил бы тот же toxic-interval, что чиним — внешняя правка без + # notify_change не была бы увидена). Дорого (~500ms), но dirty редок. + if dirty: + return _Fingerprint(root=self.root) + if self._fingerprint is not None and self._head == head: + return self._fingerprint + cached_fp = self._cache.get("fingerprint") + # ADR-0005 schema guard: кэш без "packages" (докэшовая версия) пересобирается, + # иначе пустой packages ложно REFUTED'ил бы все pkg:-якоря. + if self._cache.get("head") == head and isinstance(cached_fp, dict) and "packages" in cached_fp: + self._fingerprint = _Fingerprint(data=cached_fp) + else: + self._fingerprint = _Fingerprint(root=self.root) + self._head = head + self._cache["head"] = head + return self._fingerprint + + # ── Кэш вердиктов ── + + @staticmethod + def _cache_key(node_id: str, head: str, dirty: bool = False) -> str: + # Dirty-флаг в ключе: незакоммиченные правки не должны переиспользовать + # вердикты чистого HEAD (и наоборот) — dirty дерево = другая реальность. + return hashlib.sha256( + f"{node_id}|{head}|{int(bool(dirty))}".encode("utf-8") + ).hexdigest()[:16] + + def _load_cache(self) -> None: + try: + data = json.loads(self.cache_file.read_text(encoding="utf-8")) + if isinstance(data, dict): + self._cache = data + self._cache.setdefault("verdicts", {}) + self._cache.setdefault("counters", {}) + except (OSError, json.JSONDecodeError): + self._cache = {"head": None, "fingerprint": None, "verdicts": {}} + + def _persist_cache(self) -> None: + try: + self.cache_file.parent.mkdir(parents=True, exist_ok=True) + head = self._cache.get("head") + verdicts = { + k: v for k, v in self._cache.get("verdicts", {}).items() + if v.get("head") == head # per-node инвалидация по HEAD + } + payload = { + "head": head, + "fingerprint": self._fingerprint.to_dict() if self._fingerprint else None, + "verdicts": verdicts, + # per-node счётчики (Том): ключ node_id — переживают смену HEAD, + # в отличие от verdicts (per-head инвалидация выше). + "counters": self._cache.get("counters", {}), + } + self.cache_file.write_text( + json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8" + ) + except OSError as exc: + logger.warning("verify_on_read: cache persist failed: %s", exc) + + # ── Проверка якорей ── + + def _check_anchor(self, anchor: Anchor, fp: _Fingerprint) -> Optional[bool]: + """Проверяет якорь против отпечатка. + + Returns: + True (найден) | False (отсутствует/drift) | None (не проверяемо + — consumer обязан трактовать как INCONCLUSIVE, никогда VERIFIED). + """ + if anchor.kind == "file": + target_is_file = (self.root / anchor.value).is_file() or anchor.value in fp.files + return target_is_file + if anchor.kind == "import": + return anchor.value.split(".")[0] in fp.imports + if anchor.kind == "env": + return anchor.value in fp.env_keys + if anchor.kind == "pkg": + # ADR-0005: closed-world — манифест это источник правды для зависимостей. + return _pkg_in_pool(fp.packages, anchor.value) + if anchor.kind == "symbol": + # Referent-identity check: the qname must still resolve. Without a + # resolver (graph unavailable/not built) -> None -> INCONCLUSIVE. + if self.symbol_resolver is None: + return None + return self.symbol_resolver(anchor.value) + return None + + def _classify(self, anchors: List[Anchor], fp: _Fingerprint) -> Tuple[str, Optional[str]]: + """Вердикт: FOUND / NOT_FOUND(+проваленный якорь) / INCONCLUSIVE.""" + if not anchors: + return VERDICT_INCONCLUSIVE, None + for a in anchors: + check = self._check_anchor(a, fp) + # TRI-STATE: None = unverifiable (e.g. graph unavailable, or the + # symbol resolver could not prove index freshness) -> INCONCLUSIVE, + # NEVER promoted to VERIFIED. + if check is None: + return VERDICT_INCONCLUSIVE, None + # Symbol anchors are freshness-gated in the RESOLVER (commit B): a + # False only arrives when the index build_head matches the live HEAD + # on a clean tree, so REFUTED here is honest — no stale-index hazard. + if not check: + return VERDICT_NOT_FOUND, f"{a.kind}:{a.value}" + return VERDICT_FOUND, None + + # ── Применение переходов (общий write-путь, тот же lock) ── + + def _persist_transitions( + self, transitions: List[Dict[str, Any]], last_checked_ids: Optional[Set[str]] = None + ) -> None: + """Применяет статус-переходы и/или освежает last_checked узлов. + + last_checked_ids — узлы, реально проверенные в этом проходе (включая + INCONCLUSIVE и cache-hit): след «проверен в