From f74057f6cdb6d87b66e98967565df9fe55ccbfb8 Mon Sep 17 00:00:00 2001 From: MSCodeBase Agent Date: Sat, 26 Sep 2026 10:12:42 +0300 Subject: [PATCH] chore(experiments): save 2026-09-22 E15-E17 probes Leftover untracked probe scripts, fixture builders and one research note from the 2026-09-22 session, plus the bootstrap-pipeline blog draft and the embeddinggemma bench edits made that day. Recording them so the work is not lost. Safe ruff fixes applied (experiments/ is outside the ruff gate scope, which lints only src/ and tests/). --- docs/blog/bootstrap-pipeline.md | 68 +++++++--- experiments/bootstrap/build_gemma_tests.py | 19 +++ experiments/bootstrap/build_gemma_tests_v2.py | 33 +++++ experiments/bootstrap/check_graph_tests.py | 40 ++++++ experiments/bootstrap/debug_gemma_paths.py | 35 +++++ experiments/bootstrap/debug_get_tests.py | 60 +++++++++ experiments/bootstrap/e17_demo.py | 101 ++++++++++++++ experiments/bootstrap/e17_direct_test.py | 42 ++++++ .../bootstrap/e17_multi_project_check.py | 68 ++++++++++ experiments/bootstrap/e17_parameter_sweep.py | 120 +++++++++++++++++ .../e17_research_comparative_analysis.md | 127 ++++++++++++++++++ experiments/bootstrap/e17_smoke_answers.json | 22 +++ experiments/bootstrap/e17_smoke_prompts.json | 30 +++++ experiments/bootstrap/test_judge.py | 69 ++++++++++ experiments/embeddinggemma/_e15_probe_gold.py | 12 ++ experiments/embeddinggemma/bench.py | 30 +++-- experiments/embeddinggemma/e16_doc_corpus.py | 24 ++++ .../embeddinggemma/probe_e16_doc_corpus.py | 45 +++++++ .../embeddinggemma/probe_e16_doc_indexing.py | 39 ++++++ 19 files changed, 953 insertions(+), 31 deletions(-) create mode 100644 experiments/bootstrap/build_gemma_tests.py create mode 100644 experiments/bootstrap/build_gemma_tests_v2.py create mode 100644 experiments/bootstrap/check_graph_tests.py create mode 100644 experiments/bootstrap/debug_gemma_paths.py create mode 100644 experiments/bootstrap/debug_get_tests.py create mode 100644 experiments/bootstrap/e17_demo.py create mode 100644 experiments/bootstrap/e17_direct_test.py create mode 100644 experiments/bootstrap/e17_multi_project_check.py create mode 100644 experiments/bootstrap/e17_parameter_sweep.py create mode 100644 experiments/bootstrap/e17_research_comparative_analysis.md create mode 100644 experiments/bootstrap/e17_smoke_answers.json create mode 100644 experiments/bootstrap/e17_smoke_prompts.json create mode 100644 experiments/bootstrap/test_judge.py create mode 100644 experiments/embeddinggemma/_e15_probe_gold.py create mode 100644 experiments/embeddinggemma/e16_doc_corpus.py create mode 100644 experiments/embeddinggemma/probe_e16_doc_corpus.py create mode 100644 experiments/embeddinggemma/probe_e16_doc_indexing.py diff --git a/docs/blog/bootstrap-pipeline.md b/docs/blog/bootstrap-pipeline.md index 2ebd6cf5..98e5335e 100644 --- a/docs/blog/bootstrap-pipeline.md +++ b/docs/blog/bootstrap-pipeline.md @@ -4,7 +4,7 @@ description: "Part 4 of MSCodeBase Intelligence — Field Notes. Full source-mat tags: search, rag, codearchitecture, testing --- -> **Disclaimer & Status:** draft (source-material for the article). This is not a "feature advertisement", but an honest engineering story: figures are reproducible, weak points are named, and unaddressed risks are listed in the "What Could Go Wrong" section. +> **Disclaimer & Status:** This is an ongoing research investigation, not a feature announcement. We explore what happens when execution-derived test-to-code traceability becomes repository evidence for AI coding agents. Figures are reproducible, limitations are named, and the central question — whether this evidence helps LLMs — remains open. --- @@ -168,9 +168,9 @@ Linked % on clean (non-mocked) third-party projects proved **higher** than on ou --- -## And Now: Edges Met the Consumer (E17) +## And Now: Runtime Evidence Becomes Repository Evidence (E17) -Everything prior built `test ──TESTS──> function` edges inside PropertyGraph without leveraging them during search. E17 closed the loop: +Everything prior built `test ──TESTS──> function` edges inside PropertyGraph. The question: what happens when we expose this execution-backed evidence to the retrieval system? - **Data:** 1,727 tests → **16,172 TESTS edges**, 1,595 Test nodes, 1,132 covered functions. - **Implementation:** `SymbolIndexAdapter.get_tests_for_symbol()` (incoming `TESTS` edges) + `Searcher._append_tests_signal()`: appends up to 3 tests per function (capped at `min(len, 6)` per query), `graph_score = 0.4` vs 1.0 for definitions, sentinel `chunk_index = -(20_000_000 + line)` avoiding collisions with code chunks in RRF ranking. @@ -199,7 +199,11 @@ TESTS-signal: 34/35 queries received new covering tests (97.1%) graph_stage avg dt: off=6.52ms, on=7.53ms (overhead +15.3%) ``` -**Critical finding:** TESTS-signal **does not improve hit@1** (off=on). It only **adds context** (tests) to already-found results: 97.1% of queries received new covering tests. This means TESTS-signal is **context for LLM**, not a search improvement. If LLM doesn't use tests, the signal is useless. +**Critical finding:** TESTS-signal **does not improve retrieval ranking** (hit@1 off=on at 94.3%, MRR unchanged at 0.957). It adds execution-backed evidence to the context (97.1% of queries received new covering tests), but this evidence does not change which function is found first. + +This means the value of TESTS-signal, if any, lies **not in retrieval** but potentially in **LLM understanding**: does seeing the actual tests that exercise a function help the model understand behavior, identify edge cases, or propose safer changes? + +This remains an open experiment. ### Language Coverage (Critical Limitation) @@ -225,20 +229,18 @@ We tested TESTS-signal against 5 attack vectors: ## What Could Go Wrong -A transparent list of risks and open validation items: +### Fundamental question +The central risk is not technical but conceptual: **execution-derived test evidence may not help LLMs at all**. If models already infer behavior from code structure, naming, and docstrings, adding explicit test links may provide no additional signal. This can only be answered by measuring LLM performance with and without TESTS evidence on tasks like behavior understanding, edge case detection, and change planning. +### Technical limitations 1. **Evaluation scope.** While expanded to a 35-query panel, evaluation is still performed on a single primary codebase without deep reranker interaction. -2. **A/B did not improve hit@1.** Wide panel (35 queries): +2. **A/B did not improve retrieval.** Wide panel (35 queries): - hit@1 off=33/35 on=33/35 (94.3%) - hit@3 off=34/35 on=34/35 (97.1%) - MRR(function) off=0.957 on=0.957 - TESTS-signal **does not help find the function** (hit@1 did not improve). It only **adds context** (tests) to already-found results: 34/35 queries received new covering tests (97.1%). - - → **Conclusion:** TESTS-signal is **context for LLM**, not a search improvement. - → **Risk:** if LLM doesn't use tests, the signal is useless. - → **Fix:** verify on real LLM pipeline (not in this experiment). + TESTS-signal adds evidence but does not change retrieval ranking. The next experiment must measure LLM-level outcomes. 3. **Overhead +15.3% for wide panel.** Average graph_stage time: - off: 6.52ms @@ -246,23 +248,17 @@ A transparent list of risks and open validation items: - overhead: +15.3% For bootstrap (one-time run) this is acceptable. For prod search — may be critical with many queries. - - → **Risk:** at 1000 queries/sec, overhead may be noticeable. - → **Fix:** cache TESTS-signal (not done). 4. **Red team: 5/5 attacks repelled.** Tested: - - ✅ **Concurrency:** 10 threads × 100 calls = 1,000 calls in 17.2s, 0 errors - - ✅ **Boundaries:** function with 234 tests = 16.11ms (acceptable) - - ✅ **Abuse:** query for nonexistent function = 0 results (graceful degradation) - - ✅ **TOCTOU:** graph closed between calls = graceful degradation - - ✅ **Dependency failure:** PropertyGraph with nonexistent path = 0 results (graceful degradation) - - → **Conclusion:** TESTS-signal is resilient to concurrency, boundaries, abuse, TOCTOU, and dependency failures. + - ✅ Concurrency: 10 threads × 100 calls = 1,000 calls in 17.2s, 0 errors + - ✅ Boundaries: function with 234 tests = 16.11ms (acceptable) + - ✅ Abuse: query for nonexistent function = 0 results (graceful degradation) + - ✅ TOCTOU: graph closed between calls = graceful degradation + - ✅ Dependency failure: PropertyGraph with nonexistent path = 0 results (graceful degradation) 5. **Language limitation.** Dynamic trace is currently Python-only (~34% function coverage in Python, 0% in JS/TS/Go). 6. **`graph_score = 0.4` is an empirical constant.** Chosen to stay strictly below function definitions, but unverified against BM25/reranker weight interactions. - → Verify on full pipeline; constant may become a parameter. 7. **Pointer `:0`.** Test nodes from dynamic trace lack line numbers (`line=0`). Indexers must resolve test decorator line positions before enabling in prod. @@ -280,6 +276,34 @@ A transparent list of risks and open validation items: --- +## Related Work + +Test-to-code traceability is not a new problem. TCTracer (White & Krinke, 2022) and PyTCTracer already use dynamic execution traces to establish test→code links. Chen et al. (2025) replicated this on Python projects and found that many classical techniques work worse on Python than on Java. + +What differs in our approach is the **downstream application**: + +``` +Classical test-to-code traceability: + test → code (for maintenance, refactoring, impact analysis) + +Our approach: + test → runtime execution → persistent graph → repository retrieval → LLM context +``` + +Recent work is moving in similar directions: +- **TDAD** (2026): graph-based impact analysis for coding agents, but uses static AST +- **RepoGraph**: runtime overlay on static graph, but static-first +- **TICoder** (2026): tests as behavioral context for LLM, but without runtime trace +- **Agent Retrieval Bench** (2026): benchmark with code2test/trace2code tasks + +We have not found published work that explicitly builds the full chain from runtime execution to persistent graph to LLM context. This does not mean it doesn't exist — only that in our search we did not find it. + +Our research question is therefore: **what happens when execution-derived test-to-code traceability becomes repository evidence for an AI coding agent?** + +We do not yet know the answer. Current results show that TESTS-signal does not improve retrieval ranking (hit@1 unchanged at 94.3%). The next experiment is whether this evidence helps LLM understand behavior, identify edge cases, or propose safer changes. + +--- + ## Acknowledgments A huge thank you to everyone who engages with these posts in the comments. Your feedback, real-world observations, counter-examples, and benchmark numbers directly shape these experiments. This kind of open technical critique is what keeps engineering honest. diff --git a/experiments/bootstrap/build_gemma_tests.py b/experiments/bootstrap/build_gemma_tests.py new file mode 100644 index 00000000..35c3ae91 --- /dev/null +++ b/experiments/bootstrap/build_gemma_tests.py @@ -0,0 +1,19 @@ +# -*- coding: utf-8 -*- +"""Build TESTS edges for gemma_agent from trace_result.json.""" +import sys +from pathlib import Path + +sys.path.insert(0, r"D:\Project\MSCodeBase") +sys.stdout.reconfigure(encoding="utf-8") + +from src.core.bootstrap_tests import build_from_trace_file + +project_root = Path("D:/Project/gemma_agent") +trace_file = project_root / "trace_result.json" + +print("=" * 80) +print("Building TESTS edges for gemma_agent") +print("=" * 80) + +result = build_from_trace_file(trace_file, project_root) +print(f"\n✅ Result: {result}") diff --git a/experiments/bootstrap/build_gemma_tests_v2.py b/experiments/bootstrap/build_gemma_tests_v2.py new file mode 100644 index 00000000..0affa1e8 --- /dev/null +++ b/experiments/bootstrap/build_gemma_tests_v2.py @@ -0,0 +1,33 @@ +# -*- coding: utf-8 -*- +"""Build TESTS edges for gemma_agent with correct src_dir.""" +import sys +from pathlib import Path + +sys.path.insert(0, r"D:\Project\MSCodeBase") +sys.stdout.reconfigure(encoding="utf-8") + +import json + +from src.core.artifact_paths import get_graph_db_path +from src.core.bootstrap_tests import build_tests_edges +from src.core.graph import PropertyGraph + +project_root = Path("D:/Project/gemma_agent") +trace_file = project_root / "trace_result.json" +graph_path = get_graph_db_path(project_root) + +print("=" * 80) +print("Building TESTS edges for gemma_agent (with src_dir=core)") +print("=" * 80) + +trace = json.loads(trace_file.read_text(encoding="utf-8")) +graph_db = PropertyGraph(graph_path) + +# Try with src_dir=core +src_dir = project_root / "core" +result = build_tests_edges(trace, project_root, graph_db, src_dir=src_dir) + +print(f"\nResult: {result.as_dict()}") +print(f"Missing (first 10): {result.missing[:10]}") + +graph_db.close() diff --git a/experiments/bootstrap/check_graph_tests.py b/experiments/bootstrap/check_graph_tests.py new file mode 100644 index 00000000..80c94ac6 --- /dev/null +++ b/experiments/bootstrap/check_graph_tests.py @@ -0,0 +1,40 @@ +# -*- coding: utf-8 -*- +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +import sqlite3 + +from src.core.artifact_paths import get_graph_db_path + +db_path = get_graph_db_path(Path('.')) +conn = sqlite3.connect(str(db_path)) + +# Проверим TESTS-рёбра +print('=== TESTS edges in graph ===') +result = conn.execute("SELECT COUNT(*) FROM edges WHERE type='TESTS'").fetchone() +print(f'Total TESTS edges: {result[0]}') + +# Проверим safe_mkdir +print('\n=== safe_mkdir node ===') +result = conn.execute("SELECT id, name, file_path FROM nodes WHERE name LIKE '%safe_mkdir%' LIMIT 5").fetchall() +for r in result: + print(f' id={r[0]}, name={r[1]}, file={r[2]}') + +# Проверим TESTS-рёбра для safe_mkdir +if result: + node_id = result[0][0] + print(f'\n=== TESTS edges TO safe_mkdir (id={node_id}) ===') + edges = conn.execute(f"SELECT COUNT(*) FROM edges WHERE type='TESTS' AND target_id={node_id}").fetchone() + print(f'Incoming TESTS edges: {edges[0]}') + + # Покажем несколько тестов + tests = conn.execute(f"SELECT source_id FROM edges WHERE type='TESTS' AND target_id={node_id} LIMIT 5").fetchall() + print('\n=== Sample test nodes ===') + for t in tests: + test_node = conn.execute(f"SELECT name, file_path FROM nodes WHERE id={t[0]}").fetchone() + if test_node: + print(f' {test_node[0]} @ {test_node[1]}') + +conn.close() diff --git a/experiments/bootstrap/debug_gemma_paths.py b/experiments/bootstrap/debug_gemma_paths.py new file mode 100644 index 00000000..b6eb7ef4 --- /dev/null +++ b/experiments/bootstrap/debug_gemma_paths.py @@ -0,0 +1,35 @@ +# -*- coding: utf-8 -*- +"""Debug: check how functions are stored in gemma_agent graph.""" +import sys +from pathlib import Path + +sys.path.insert(0, r"D:\Project\MSCodeBase") +sys.stdout.reconfigure(encoding="utf-8") + +from src.core.artifact_paths import get_graph_db_path +from src.core.graph import NodeLabel, PropertyGraph + +project_root = Path("D:/Project/gemma_agent") +graph_path = get_graph_db_path(project_root) +graph_db = PropertyGraph(graph_path) + +# Check sample function nodes +print("=== Sample Function nodes in graph ===") +funcs = graph_db.find_nodes(label=NodeLabel.FUNCTION, limit=10) +for f in funcs: + print(f" name={f.name}, file_path={f.file_path}") + +# Check trace paths +print("\n=== Sample trace entries ===") +import json + +trace_file = project_root / "trace_result.json" +trace = json.loads(trace_file.read_text(encoding="utf-8")) +for i, (nodeid, entries) in enumerate(trace.items()): + if i >= 3: + break + print(f" {nodeid}") + for e in entries[:3]: + print(f" {e}") + +graph_db.close() diff --git a/experiments/bootstrap/debug_get_tests.py b/experiments/bootstrap/debug_get_tests.py new file mode 100644 index 00000000..76feb640 --- /dev/null +++ b/experiments/bootstrap/debug_get_tests.py @@ -0,0 +1,60 @@ +# -*- coding: utf-8 -*- +"""Debug get_tests_for_symbol step by step.""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +from src.core.artifact_paths import get_graph_db_path +from src.core.graph import EdgeType, NodeLabel, PropertyGraph +from src.core.search.graph_adapter import SymbolIndexAdapter + +ROOT = Path(__file__).resolve().parents[2] +pg = PropertyGraph(get_graph_db_path(ROOT)) +adapter = SymbolIndexAdapter(pg, mode=SymbolIndexAdapter.MODE_PURE) + +symbol = "safe_mkdir" +file_path = "src/core/artifact_paths.py" + +print("=== Debug get_tests_for_symbol ===") +print(f"symbol: {symbol}") +print(f"file_path: {file_path}") + +# Step 1: find_nodes +print(f"\n[Step 1] find_nodes(label=FUNCTION, name_pattern=%{symbol}%, file_path={file_path})") +candidates = pg.find_nodes( + label=NodeLabel.FUNCTION, + name_pattern=f"%{symbol}%", + file_path=file_path, + limit=5, +) +print(f" Found {len(candidates)} candidates") +for c in candidates: + print(f" {c.name} @ {c.file_path} (qname={c.qualified_name})") + +# Step 2: Попробуем без file_path +print("\n[Step 2] find_nodes без file_path") +candidates2 = pg.find_nodes( + label=NodeLabel.FUNCTION, + name_pattern=f"%{symbol}%", + limit=5, +) +print(f" Found {len(candidates2)} candidates") +for c in candidates2: + print(f" {c.name} @ {c.file_path}") + +# Step 3: Проверим get_neighbors для первого кандидата +if candidates2: + node = candidates2[0] + print(f"\n[Step 3] get_neighbors({node.qualified_name}, TESTS, incoming)") + neighbors = pg.get_neighbors( + node.qualified_name, + edge_type=EdgeType.TESTS, + direction="incoming", + max_nodes=10, + ) + print(f" Found {len(neighbors)} neighbors") + for n, e, d in neighbors[:5]: + print(f" {n.name} @ {n.file_path} (label={n.label})") + +pg.close() diff --git a/experiments/bootstrap/e17_demo.py b/experiments/bootstrap/e17_demo.py new file mode 100644 index 00000000..883df9c1 --- /dev/null +++ b/experiments/bootstrap/e17_demo.py @@ -0,0 +1,101 @@ +# -*- coding: utf-8 -*- +"""E17 Demo: show what LLM receives WITHOUT vs WITH TESTS evidence. + +Цель: показать ЦЕННОСТЬ TESTS-signal на реальном примере. +""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) +from src.core.artifact_paths import get_graph_db_path +from src.core.graph import PropertyGraph +from src.core.search.engine import Searcher +from src.core.search.graph_adapter import SymbolIndexAdapter + +sys.stdout.reconfigure(encoding="utf-8") +ROOT = Path(__file__).resolve().parents[2] + + +class FakeIndexer: + def __init__(self, symbol_index): + self._symbol_index = symbol_index + self.symbol_index = symbol_index + async def close_async(self): + pass + + +def demo_query(query: str): + """Показывает что получает LLM без TESTS и с TESTS.""" + pg = PropertyGraph(get_graph_db_path(ROOT)) + adapter = SymbolIndexAdapter(pg, mode=SymbolIndexAdapter.MODE_PURE) + searcher = Searcher(indexer=FakeIndexer(adapter), embedder=None) + + print("=" * 80) + print(f"QUERY: {query}") + print("=" * 80) + + # WITHOUT TESTS + searcher._tests_signal = False + out_off = searcher._graph_stage(query, limit=5) + + print("\n[WITHOUT TESTS] Что получает LLM:") + print("-" * 80) + for i, r in enumerate(out_off[:3], 1): + meta = r.get("metadata", {}) + print(f"{i}. {meta.get('symbol')} @ {meta.get('file')}:{meta.get('line')}") + print(f" score: {r.get('final_score', 0):.3f}") + + # WITH TESTS + searcher._tests_signal = True + out_on = searcher._graph_stage(query, limit=5) + + print("\n[WITH TESTS] Что получает LLM:") + print("-" * 80) + for i, r in enumerate(out_on[:5], 1): + meta = r.get("metadata", {}) + symbol = meta.get('symbol') + file = meta.get('file') + line = meta.get('line') + score = r.get('final_score', 0) + is_test = meta.get('tests_signal', False) + covers = meta.get('covers', '') + + if is_test: + print(f"{i}. 🧪 TEST: {symbol}") + print(f" covers: {covers}") + print(f" file: {file}:{line}") + else: + print(f"{i}. 📄 FUNCTION: {symbol}") + print(f" file: {file}:{line}") + print(f" score: {score:.3f}") + + # Анализ + tests_added = [r for r in out_on if r.get("metadata", {}).get("tests_signal")] + print(f"\n📊 Добавлено тестов: {len(tests_added)}") + if tests_added: + print(" LLM теперь видит:") + for t in tests_added[:3]: + meta = t.get("metadata", {}) + print(f" • {meta.get('symbol')} → проверяет {meta.get('covers')}") + + pg.close() + + +def main(): + print("\n" + "=" * 80) + print("E17 DEMONSTRATION: TESTS Evidence Value") + print("=" * 80) + print("\nПоказываю что РЕАЛЬНО получает LLM с TESTS-signal и без него.\n") + + # Реальные запросы + demo_query("safe_mkdir") + print("\n\n") + + demo_query("get_project_dir") + print("\n\n") + + demo_query("PropertyGraph") + + +if __name__ == "__main__": + main() diff --git a/experiments/bootstrap/e17_direct_test.py b/experiments/bootstrap/e17_direct_test.py new file mode 100644 index 00000000..cd2a2035 --- /dev/null +++ b/experiments/bootstrap/e17_direct_test.py @@ -0,0 +1,42 @@ +# -*- coding: utf-8 -*- +"""E17 Direct test: get_tests_for_symbol with different limits.""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) +from src.core.artifact_paths import get_graph_db_path +from src.core.graph import PropertyGraph +from src.core.search.graph_adapter import SymbolIndexAdapter + +sys.stdout.reconfigure(encoding="utf-8") +ROOT = Path(__file__).resolve().parents[2] + + +def main(): + pg = PropertyGraph(get_graph_db_path(ROOT)) + adapter = SymbolIndexAdapter(pg, mode=SymbolIndexAdapter.MODE_PURE) + + # Тестовые функции с разным количеством TESTS-рёбер + test_cases = [ + ("safe_mkdir", "src/core/artifact_paths.py"), # 234 теста (hub) + ("get_project_dir", "src/core/artifact_paths.py"), # много тестов + ("_graph_stage", "src/core/search/engine.py"), # мало тестов + ] + + print("=" * 80) + print("DIRECT TEST: get_tests_for_symbol with different limits") + print("=" * 80) + + for symbol, file_path in test_cases: + print(f"\n[{symbol}] @ {file_path}") + + for limit in [1, 2, 3, 5, 10]: + tests = adapter.get_tests_for_symbol(symbol, file_path, limit=limit) + test_names = [t.symbol for t in tests[:5]] + print(f" limit={limit:2d}: got {len(tests):2d} tests → {test_names}") + + pg.close() + + +if __name__ == "__main__": + main() diff --git a/experiments/bootstrap/e17_multi_project_check.py b/experiments/bootstrap/e17_multi_project_check.py new file mode 100644 index 00000000..977e9835 --- /dev/null +++ b/experiments/bootstrap/e17_multi_project_check.py @@ -0,0 +1,68 @@ +# -*- coding: utf-8 -*- +"""E17 Multi-project check: which projects have PropertyGraph + TESTS edges.""" +import sys +from pathlib import Path + +sys.path.insert(0, r"D:\Project\MSCodeBase") +sys.stdout.reconfigure(encoding="utf-8") + +import sqlite3 + +from src.core.artifact_paths import get_graph_db_path + +projects = [ + "gemma_agent", + "codebase-memory-mcp-main", + "commit-", + "MSPortfolio", + "OpenCodeClient", + "Bot_snow", + "TorrServer-master", + "MSCodeBase", +] + +print("=" * 80) +print("Multi-project PropertyGraph check") +print("=" * 80) + +for proj_name in projects: + proj_path = Path(f"D:/Project/{proj_name}") + if not proj_path.exists(): + print(f"\n❌ {proj_name}: not found") + continue + + try: + graph_path = get_graph_db_path(proj_path) + if not graph_path.exists(): + print(f"\n⚠️ {proj_name}: no PropertyGraph (not indexed)") + continue + + conn = sqlite3.connect(str(graph_path)) + + # Count nodes by type + funcs = conn.execute("SELECT COUNT(*) FROM nodes WHERE label='Function'").fetchone()[0] + tests = conn.execute("SELECT COUNT(*) FROM nodes WHERE label='Test'").fetchone()[0] + tests_edges = conn.execute("SELECT COUNT(*) FROM edges WHERE type='TESTS'").fetchone()[0] + + # Language coverage + py_funcs = conn.execute("SELECT COUNT(*) FROM nodes WHERE label='Function' AND file_path LIKE '%.py'").fetchone()[0] + ts_funcs = conn.execute("SELECT COUNT(*) FROM nodes WHERE label='Function' AND (file_path LIKE '%.ts' OR file_path LIKE '%.tsx')").fetchone()[0] + go_funcs = conn.execute("SELECT COUNT(*) FROM nodes WHERE label='Function' AND file_path LIKE '%.go'").fetchone()[0] + + # Functions with TESTS edges + funcs_with_tests = conn.execute(""" + SELECT COUNT(DISTINCT target_id) + FROM edges + WHERE type='TESTS' + """).fetchone()[0] + + conn.close() + + print(f"\n✅ {proj_name}") + print(f" Functions: {funcs} (Python: {py_funcs}, TS: {ts_funcs}, Go: {go_funcs})") + print(f" Test nodes: {tests}") + print(f" TESTS edges: {tests_edges}") + print(f" Functions with TESTS: {funcs_with_tests} ({funcs_with_tests/funcs*100:.1f}%)" if funcs > 0 else " Functions with TESTS: 0") + + except Exception as e: + print(f"\n❌ {proj_name}: error - {e}") diff --git a/experiments/bootstrap/e17_parameter_sweep.py b/experiments/bootstrap/e17_parameter_sweep.py new file mode 100644 index 00000000..8d092274 --- /dev/null +++ b/experiments/bootstrap/e17_parameter_sweep.py @@ -0,0 +1,120 @@ +# -*- coding: utf-8 -*- +"""E17 Parameter Sweep: graph_score, test_limit variations. + +Цель: понять как параметры влияют на результаты. +""" +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) +from src.core.artifact_paths import get_graph_db_path +from src.core.graph import PropertyGraph +from src.core.search.engine import Searcher +from src.core.search.graph_adapter import SymbolIndexAdapter + +sys.stdout.reconfigure(encoding="utf-8") +ROOT = Path(__file__).resolve().parents[2] + + +class FakeIndexer: + def __init__(self, symbol_index): + self._symbol_index = symbol_index + self.symbol_index = symbol_index + async def close_async(self): + pass + + +# 10 реальных запросов (смешанные: identifier + NL-like) +QUERIES = [ + "safe_mkdir", + "get_project_dir", + "PropertyGraph", + "Searcher", + "how does reindex work", + "embedding model config", + "graph_stage", + "TESTS edges", + "bootstrap pipeline", + "artifact_paths", +] + + +def run_sweep(graph_score: float, test_limit_per_func: int, test_limit_per_query: int): + """Прогоняет панель с заданными параметрами.""" + pg = PropertyGraph(get_graph_db_path(ROOT)) + adapter = SymbolIndexAdapter(pg, mode=SymbolIndexAdapter.MODE_PURE) + searcher = Searcher(indexer=FakeIndexer(adapter), embedder=None) + searcher._tests_signal = True + + results = [] + for q in QUERIES: + out = searcher._graph_stage(q, limit=8) + + # Считаем тесты в результате + tests_in_result = [r for r in out if r.get("metadata", {}).get("tests_signal")] + funcs_in_result = [r for r in out if r.get("metadata", {}).get("is_definition")] + + results.append({ + "query": q, + "total_results": len(out), + "funcs": len(funcs_in_result), + "tests": len(tests_in_result), + "test_symbols": [r.get("metadata", {}).get("symbol") for r in tests_in_result[:3]], + }) + + pg.close() + return results + + +def main(): + print("=" * 80) + print("E17 PARAMETER SWEEP") + print("=" * 80) + + # Базовый вариант (текущие параметры) + print("\n[BASELINE] graph_score=0.4, limit=3 per func, cap=6 per query") + baseline = run_sweep(0.4, 3, 6) + total_tests_baseline = sum(r["tests"] for r in baseline) + print(f" Total tests added: {total_tests_baseline}") + for r in baseline[:3]: + print(f" {r['query']}: {r['funcs']} funcs, {r['tests']} tests") + + # Вариант 1: выше graph_score + print("\n[VARIANT 1] graph_score=0.6 (выше)") + v1 = run_sweep(0.6, 3, 6) + total_tests_v1 = sum(r["tests"] for r in v1) + print(f" Total tests added: {total_tests_v1}") + + # Вариант 2: ниже graph_score + print("\n[VARIANT 2] graph_score=0.2 (ниже)") + v2 = run_sweep(0.2, 3, 6) + total_tests_v2 = sum(r["tests"] for r in v2) + print(f" Total tests added: {total_tests_v2}") + + # Вариант 3: больше тестов на функцию + print("\n[VARIANT 3] limit=5 per func (больше)") + v3 = run_sweep(0.4, 5, 6) + total_tests_v3 = sum(r["tests"] for r in v3) + print(f" Total tests added: {total_tests_v3}") + + # Вариант 4: меньше тестов на функцию + print("\n[VARIANT 4] limit=1 per func (меньше)") + v4 = run_sweep(0.4, 1, 6) + total_tests_v4 = sum(r["tests"] for r in v4) + print(f" Total tests added: {total_tests_v4}") + + print("\n" + "=" * 80) + print("SUMMARY") + print("=" * 80) + print(f"Baseline (0.4, limit=3): {total_tests_baseline} tests") + print(f"Higher score (0.6): {total_tests_v1} tests") + print(f"Lower score (0.2): {total_tests_v2} tests") + print(f"More per func (limit=5): {total_tests_v3} tests") + print(f"Less per func (limit=1): {total_tests_v4} tests") + + print("\nNOTE: graph_score не влияет на количество тестов (они добавляются отдельно).") + print(" graph_score влияет только на ранжирование в RRF с другими сигналами.") + + +if __name__ == "__main__": + main() diff --git a/experiments/bootstrap/e17_research_comparative_analysis.md b/experiments/bootstrap/e17_research_comparative_analysis.md new file mode 100644 index 00000000..acf3f4a4 --- /dev/null +++ b/experiments/bootstrap/e17_research_comparative_analysis.md @@ -0,0 +1,127 @@ +# E17 Research: Comparative Analysis + +## Сравнительная таблица: Test-to-Code Traceability → LLM Context + +| Работа | Год | Язык | Механика | Данные | Доказательство | Хранение | Применение | Измерение | +|--------|-----|------|----------|--------|----------------|----------|------------|-----------| +| **TCTracer** | 2022 | Java | Dynamic trace + static ensemble | Call traces, naming, TF-IDF | MAP 85% (method), 92% (class) | Links file | Maintenance, refactoring | Precision/Recall/MAP | +| **PyTCTracer** | 2022 | Python | sys.settrace + pytest plugin | CSV trace log | Same techniques as TCTracer | JSON links | Traceability recovery | Precision/Recall/F1/MAP | +| **Chen et al.** | 2025 | Python | 15 техник (dynamic + static) | 7 projects, 3198 tests | Cross-level info | N/A | Traceability techniques | Recall/Precision trade-offs | +| **TDAD** | 2026 | Python | Static AST → code-test graph | Dependency map | Regressions -70% (6.08%→1.82%) | Agent skill (text file) | Pre-change impact analysis | Regression rate, resolution rate | +| **RepoGraph** | 2026 | Py/JS/TS | Static-first + runtime overlay | Graph DB (Kuzu) | Pathway scoring | Graph DB | Repository intelligence | Qualitative (pathways, dead code) | +| **TICoder** | 2026 | Multi | Tests as behavioral specs | Test cases | +11.52% on benchmarks | N/A | Code generation planning | Pass@k on benchmarks | +| **ARB** | 2026 | Multi | Benchmark: code2test, trace2code | 427 samples, 25 repos | MRR/Recall/BCY@8k | Dataset | Retrieval evaluation | MRR, Recall@20, BCY@8k | +| **Syncause** | 2026 | Python | Runtime tracing → context injection | Call traces | +6% on SWE-bench (77.4%→83.4%) | Runtime facts | LLM context injection | SWE-bench score | +| **Shiplight** | 2026 | Multi | Agent-first testing | Browser traces | Qualitative | YAML tests | Agent verification | Qualitative (agent workflow) | +| **CKG** | 2026 | Multi | Code Knowledge Graph | AST, imports, calls | Graph queries | SQLite + Rust | MCP retrieval | Qualitative | +| **ABCoder** | 2026 | TS | Graph-based indexing | UniAST | Function-level retrieval | Graph index | Code agent context | Retrieval accuracy | +| **DeepDiscovery** | 2026 | Multi | Task-level context recovery | Multi-relational graph | Implementation path recovery | Structured context | Downstream reasoning | SWE performance | +| **E17 (наш)** | 2026 | Python | Runtime trace → TESTS edges → search | sys.settrace, PropertyGraph | 97.1% queries get tests, retrieval unchanged | PropertyGraph (SQLite) | Search context | hit@1/MRR (unchanged) | + +## Ключевые различия + +### TCTracer / PyTCTracer / Chen et al. +**Задача:** Восстановить test-to-code traceability links +**Цель:** Maintenance, refactoring, impact analysis +**Результат:** Links (test → code) +**НЕ делают:** Не используют links для LLM context + +### TDAD +**Задача:** Уменьшить regressions от AI coding agents +**Механика:** Static AST → dependency map → agent skill +**Результат:** -70% regressions +**Отличие от нас:** Static (не runtime), не persistent graph + +### RepoGraph +**Задача:** Repository intelligence +**Механика:** Static-first + runtime overlay +**Результат:** Pathways, dead code, variable flows +**Отличие от нас:** Не специализируется на test-to-code, runtime = overlay (не evidence) + +### TICoder +**Задача:** Repository-level code generation +**Механика:** Tests as behavioral specs → planning +**Результат:** +11.52% on benchmarks +**Отличие от нас:** Tests как specs (не execution evidence), не persistent graph + +### ARB +**Задача:** Benchmark для retrieval +**Механика:** code2test, trace2code задачи +**Результат:** Leaderboard (MRR, Recall) +**Отличие от нас:** Benchmark (не система), не использует runtime trace + +### Syncause ⚠️ КРИТИЧЕСКИЙ КОНКУРЕНТ +**Задача:** Inject runtime context в LLM +**Механика:** Runtime tracing → facts → LLM context +**Результат:** +6% на SWE-bench (77.4% → 83.4%) +**Отличие от нас:** +- Не persistent graph (runtime facts per query) +- Не test-to-code specifically (general runtime tracing) +- Не PropertyGraph с TESTS edges + +### Наш E17 +**Задача:** Execution-backed test evidence для LLM +**Механика:** Runtime trace → TESTS edges → persistent graph → search context +**Результат:** 97.1% queries get tests, retrieval unchanged +**Уникальность:** +1. Persistent TESTS edges в PropertyGraph (не per-query) +2. TESTS-signal в search (не injection) +3. Execution-backed (не static, не behavioral specs) + +**НО:** +- Retrieval не улучшился (hit@1 unchanged) +- LLM impact НЕ измерен +- Syncause показал +6% на SWE-bench с runtime tracing + +## Выводы + +### Что НЕ ново: +- test-to-code traceability (TCTracer 2022) +- Dynamic execution traces для linking (PyTCTracer) +- Tests как context для LLM (TICoder, Shiplight) +- Runtime tracing для LLM context (Syncause) +- Graph-based code indexing (ABCoder, CKG, RepoGraph) + +### Что может быть ново: +1. **Persistent TESTS edges в PropertyGraph** — не per-query runtime facts, а permanent graph structure +2. **TESTS-signal в search** — не injection, а часть retrieval pipeline +3. **Полная цепочка:** runtime trace → persistent graph → search → LLM context + +### Но: +- **Syncause** делает похожее (runtime → LLM) и показывает +6% +- **Мы не измерили LLM impact** — retrieval unchanged, но LLM? +- **Наша уникальность не в результате, а в архитектуре** (persistent vs per-query) + +### Следующий эксперимент (критический): +Измерить LLM-level impact: +- 30 вопросов к кодовой базе +- A/B: без TESTS vs с TESTS +- Метрики: behavior understanding, edge case detection, change planning + +Если LLM impact = 0 → инфраструктура работает, но ценность не доказана +Если LLM impact > 0 → уникальная архитектура (persistent TESTS edges) имеет значение + +## Дополнительные находки + +### Reddit r/AI_Agents (2026-03-04): +"We added runtime tracing to an SWE-bench agent and pushed Gemini 3 Pro from 77.4% to 83.4%" +- Syncause: runtime facts → LLM context → +6% +- Подтверждает гипотезу: runtime evidence помогает LLM + +### Reddit r/LLMDevs (2026-06-07): +"What if agent traces became a behavior graph?" +- Trajectory failures ≠ answer-quality issues +- Behavior graph для agent traces +- Похожая идея: traces → graph structure + +### DEV.to (2026-04-10): +"Agent-First Testing: Build Quality Into Every AI Coding Session" +- Shiplight: agent writes tests during development +- Tests как verification, не как evidence +- Другое применение (не retrieval) + +### DEV.to (2026-08-01): +"Keeping Specs, Tests, And Code In Sync In AI Development" +- Traceability model: requirement → design → test → code +- Spec-to-test mapping (не execution-backed) +- Static traceability (не runtime) diff --git a/experiments/bootstrap/e17_smoke_answers.json b/experiments/bootstrap/e17_smoke_answers.json new file mode 100644 index 00000000..cfdb43c4 --- /dev/null +++ b/experiments/bootstrap/e17_smoke_answers.json @@ -0,0 +1,22 @@ +[ + { + "id": 13, + "mode": "A_baseline", + "answer": "`tokenize` converts the Cypher query string into a list of `Token` objects. It takes no explicit arguments, reads `self._query`, and walks it character by character using index `i`, appending `Token(type, text, position)` entries; it returns `List[Token]`. It skips whitespace, discards `//` line comments (jumping to the next newline), and recognizes quoted strings in single or double quotes, handling backslash escapes. Unterminated strings raise `SyntaxError`. Numbers are scanned as digits plus dots, including a leading-dot decimal. Identifiers, keywords, and backtick-escaped identifiers are recognized starting with a letter, underscore, or backtick. Edge cases: a comment without a trailing newline stops tokenization via `break`; an escaped character inside a string is skipped by advancing twice; number scanning accepts multiple dots because it only tests `isdigit()` or membership in `.`; backtick handling searches for the closing backtick." + }, + { + "id": 13, + "mode": "B_runtime", + "answer": "`tokenize` turns `self._query` into a `List[Token]`, walking the string with an index `i` and emitting tokens carrying type, text, and start position. Whitespace is skipped; `//` comments are dropped by seeking the next newline; single- and double-quoted strings are captured with backslash escapes, and an unterminated string raises `SyntaxError`. Numeric literals (digits, optionally with a leading dot) are consumed greedily. Identifiers, keywords, and backtick-quoted identifiers are recognized. Because the runtime context includes tests exercising related storage paths, the emphasis is on observed behavior rather than just structure, but the visible logic is the same: edge cases are EOF-terminated comments (`break`), escape sequences skipping a character, multi-dot number runs, and a missing closing backtick. Output is deterministic for a given query, with positions recorded relative to the original string." + }, + { + "id": 1, + "mode": "A_baseline", + "answer": "`_ensure_data_root` guarantees a usable data-root directory. It receives a `Path` named `root` and returns a `Path`. It first calls `safe_mkdir(root, what=\"data root\")`; if that succeeds it returns `root` unchanged. If `safe_mkdir` raises `ArtifactStorageError` — for example a full disk or broken permissions — the function imports `tempfile`, creates a fallback directory with `tempfile.mkdtemp(prefix=\"mscodebase_data_\")`, logs a warning containing the original root and the error, and returns the fallback path. Its purpose is resilience: the server must not crash when storage is unavailable, so it degrades to temporary storage and keeps running. Edge cases: exceptions other than `ArtifactStorageError` are not caught and propagate; the fallback is genuinely temporary and its contents may be lost; callers cannot distinguish the real root from the fallback without comparing paths; `tempfile` is imported lazily only in the failure branch." + }, + { + "id": 1, + "mode": "B_runtime", + "answer": "`_ensure_data_root` ensures the data root exists, taking a `Path` and returning a `Path`. It attempts `safe_mkdir(root, what=\"data root\")` and returns `root` on success. On `ArtifactStorageError` it creates a `tempfile.mkdtemp(prefix=\"mscodebase_data_\")` fallback, logs a warning, and returns that path so the server keeps running despite unavailable storage. The accompanying tests exercise `ActionReceiptStore` over a valid `tmp_path`, checking that `record` succeeds, `path` exists, `get` returns the stored receipt and its verdict, unknown IDs return `None`, `query(limit=...)` returns the latest entries, `count` is correct, and re-verification makes `get` return the superseding receipt. These confirm the store behaves correctly when the root is writable. Edge cases remain: non-`ArtifactStorageError` exceptions propagate, the fallback is temporary, and callers cannot tell fallback from real root without comparison." + } +] diff --git a/experiments/bootstrap/e17_smoke_prompts.json b/experiments/bootstrap/e17_smoke_prompts.json new file mode 100644 index 00000000..5a0e887a --- /dev/null +++ b/experiments/bootstrap/e17_smoke_prompts.json @@ -0,0 +1,30 @@ +[ + { + "id": 13, + "name": "CypherLexer.tokenize", + "mode": "A_baseline", + "question": "What does the function `CypherLexer.tokenize` in `cypher_lexer.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.", + "prompt": "You are an expert Python developer. Answer the following question about a function in a codebase.\n\nQuestion: What does the function `CypherLexer.tokenize` in `cypher_lexer.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.\n\nFunction code:\n```python\ndef tokenize(self) -> List[Token]:\n \"\"\"Разбивает строку запроса на токены.\"\"\"\n tokens: List[Token] = []\n i = 0\n q = self._query\n\n while i < len(q):\n ch = q[i]\n\n # Пропускаем пробелы\n if ch in \" \\t\\n\\r\":\n i += 1\n continue\n\n # Комментарии //\n if ch == \"/\" and i + 1 < len(q) and q[i + 1] == \"/\":\n end = q.find(\"\\n\", i)\n if end == -1:\n break\n i = end + 1\n continue\n\n # Строки в кавычках\n if ch in \"\\\"'\":\n end = ch\n j = i + 1\n while j < len(q) and q[j] != end:\n if q[j] == \"\\\\\":\n j += 1\n j += 1\n if j >= len(q):\n raise SyntaxError(f\"Unterminated string at pos {i}\")\n tokens.append(Token(TokenType.STRING, q[i + 1:j], i))\n i = j + 1\n continue\n\n # Числа\n if ch.isdigit() or (ch == \".\" and i + 1 < len(q) and q[i + 1].isdigit()):\n j = i\n while j < len(q) and (q[j].isdigit() or q[j] in \".\"):\n j += 1\n tokens.append(Token(TokenType.NUMBER, q[i:j], i))\n i = j\n continue\n\n # Идентификаторы, ключевые слова, метки\n if ch.isalpha() or ch == \"_\" or ch == \"`\":\n if ch == \"`\":\n # Backtick-escaped identifier\n j = q.find(\"`\", i + 1)\n if j == -1:\n raise SyntaxError(f\"Unterminated backtick at pos {i}\")\n tokens.append(Token(TokenType.IDENTIFIER, q[i + 1:j], i))\n i = j + 1\n continue\n\n j = i\n while j < len(q) and (q[j].isalnum() or q[j] == \"_\"):\n j += 1\n word = q[i:j]\n upper = word.upper()\n\n if upper in KEYWORDS:\n tt = TokenType.KEYWORD\n elif word[0].isupper() and \":\" not in word:\n tt = TokenType.LABEL\n else:\n tt = TokenType.IDENTIFIER\n\n tokens.append(Token(tt, word, i))\n i = j\n continue\n\n # Операторы сравнения (двухсимвольные)\n if i + 1 < len(q) and q[i:i + 2] in (\">=\", \"<=\", \"<>\", \"!=\", \"=~\"):\n tokens.append(Token(TokenType.OPERATOR, q[i:i + 2], i))\n i += 2\n continue\n\n # Стрелки направлений\n if i + 2 < len(q) and q[i:i + 2] == \"<-\":\n tokens.append(Token(TokenType.PUNCTUATION, \"<-\", i))\n i += 2\n continue\n if i + 1 < len(q) and q[i:i + 2] == \"->\":\n tokens.append(Token(TokenType.PUNCTUATION, \"->\", i))\n i += 2\n continue\n if i + 1 < len(q) and q[i:i + 2] == \"--\":\n tokens.append(Token(TokenType.PUNCTUATION, \"--\", i))\n i += 2\n continue\n\n # Односимвольные операторы\n if ch in \"=<>!+-*/%\":\n tokens.append(Token(TokenType.OPERATOR, ch, i))\n i += 1\n continue\n\n # Скобки и пунктуация\n if ch in \"()[]{}.,*\":\n tokens.append(Token(TokenType.PUNCTUATION, ch, i))\n i += 1\n continue\n\n # Метка :Label или :TYPE или :TYPE*min..max\n if ch == \":\":\n j = i + 1\n while j < len(q) and (q[j].isalnum() or q[j] in \"_\"):\n j += 1\n\n # Проверяем на variable-length path\n label = q[i + 1:j]\n rest = q[j:j + 10] # смотрим вперед на *min..max\n\n if rest.startswith(\"*\"):\n # :TYPE*min..max или :TYPE*\n k = j + 1\n while k < len(q) and (q[k].isdigit() or q[k] in \"..\"):\n k += 1\n path_range = q[j:k] # *1..3 или *\n tokens.append(Token(TokenType.REL_TYPE, f\"{label}{path_range}\", i))\n i = k\n else:\n tokens.append(Token(TokenType.LABEL, label if label else \"\", i))\n i = j\n continue\n\n # Неизвестный символ\n raise SyntaxError(f\"Unexpected character '{ch}' at pos {i}\")\n\n self._tokens = tokens\n return tokens\n```\n\nProvide a clear, accurate explanation of what the function does, its purpose, inputs, outputs, and any edge cases it handles." + }, + { + "id": 13, + "name": "CypherLexer.tokenize", + "mode": "B_runtime", + "question": "What does the function `CypherLexer.tokenize` in `cypher_lexer.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.", + "prompt": "You are an expert Python developer. Answer the following question about a function in a codebase.\n\nQuestion: What does the function `CypherLexer.tokenize` in `cypher_lexer.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.\n\nFunction code:\n```python\ndef tokenize(self) -> List[Token]:\n \"\"\"Разбивает строку запроса на токены.\"\"\"\n tokens: List[Token] = []\n i = 0\n q = self._query\n\n while i < len(q):\n ch = q[i]\n\n # Пропускаем пробелы\n if ch in \" \\t\\n\\r\":\n i += 1\n continue\n\n # Комментарии //\n if ch == \"/\" and i + 1 < len(q) and q[i + 1] == \"/\":\n end = q.find(\"\\n\", i)\n if end == -1:\n break\n i = end + 1\n continue\n\n # Строки в кавычках\n if ch in \"\\\"'\":\n end = ch\n j = i + 1\n while j < len(q) and q[j] != end:\n if q[j] == \"\\\\\":\n j += 1\n j += 1\n if j >= len(q):\n raise SyntaxError(f\"Unterminated string at pos {i}\")\n tokens.append(Token(TokenType.STRING, q[i + 1:j], i))\n i = j + 1\n continue\n\n # Числа\n if ch.isdigit() or (ch == \".\" and i + 1 < len(q) and q[i + 1].isdigit()):\n j = i\n while j < len(q) and (q[j].isdigit() or q[j] in \".\"):\n j += 1\n tokens.append(Token(TokenType.NUMBER, q[i:j], i))\n i = j\n continue\n\n # Идентификаторы, ключевые слова, метки\n if ch.isalpha() or ch == \"_\" or ch == \"`\":\n if ch == \"`\":\n # Backtick-escaped identifier\n j = q.find(\"`\", i + 1)\n if j == -1:\n raise SyntaxError(f\"Unterminated backtick at pos {i}\")\n tokens.append(Token(TokenType.IDENTIFIER, q[i + 1:j], i))\n i = j + 1\n continue\n\n j = i\n while j < len(q) and (q[j].isalnum() or q[j] == \"_\"):\n j += 1\n word = q[i:j]\n upper = word.upper()\n\n if upper in KEYWORDS:\n tt = TokenType.KEYWORD\n elif word[0].isupper() and \":\" not in word:\n tt = TokenType.LABEL\n else:\n tt = TokenType.IDENTIFIER\n\n tokens.append(Token(tt, word, i))\n i = j\n continue\n\n # Операторы сравнения (двухсимвольные)\n if i + 1 < len(q) and q[i:i + 2] in (\">=\", \"<=\", \"<>\", \"!=\", \"=~\"):\n tokens.append(Token(TokenType.OPERATOR, q[i:i + 2], i))\n i += 2\n continue\n\n # Стрелки направлений\n if i + 2 < len(q) and q[i:i + 2] == \"<-\":\n tokens.append(Token(TokenType.PUNCTUATION, \"<-\", i))\n i += 2\n continue\n if i + 1 < len(q) and q[i:i + 2] == \"->\":\n tokens.append(Token(TokenType.PUNCTUATION, \"->\", i))\n i += 2\n continue\n if i + 1 < len(q) and q[i:i + 2] == \"--\":\n tokens.append(Token(TokenType.PUNCTUATION, \"--\", i))\n i += 2\n continue\n\n # Односимвольные операторы\n if ch in \"=<>!+-*/%\":\n tokens.append(Token(TokenType.OPERATOR, ch, i))\n i += 1\n continue\n\n # Скобки и пунктуация\n if ch in \"()[]{}.,*\":\n tokens.append(Token(TokenType.PUNCTUATION, ch, i))\n i += 1\n continue\n\n # Метка :Label или :TYPE или :TYPE*min..max\n if ch == \":\":\n j = i + 1\n while j < len(q) and (q[j].isalnum() or q[j] in \"_\"):\n j += 1\n\n # Проверяем на variable-length path\n label = q[i + 1:j]\n rest = q[j:j + 10] # смотрим вперед на *min..max\n\n if rest.startswith(\"*\"):\n # :TYPE*min..max или :TYPE*\n k = j + 1\n while k < len(q) and (q[k].isdigit() or q[k] in \"..\"):\n k += 1\n path_range = q[j:k] # *1..3 или *\n tokens.append(Token(TokenType.REL_TYPE, f\"{label}{path_range}\", i))\n i = k\n else:\n tokens.append(Token(TokenType.LABEL, label if label else \"\", i))\n i = j\n continue\n\n # Неизвестный символ\n raise SyntaxError(f\"Unexpected character '{ch}' at pos {i}\")\n\n self._tokens = tokens\n return tokens\n```\n\nRelevant tests that exercise this function:\n\n```python\ndef test_tokenize_simple_match(self):\n tokens = CypherLexer(\"MATCH (n:Function) RETURN n.name\").tokenize()\n assert len(tokens) > 0\n types = [t.type for t in tokens]\n assert TokenType.KEYWORD in types\n```\n\n```python\ndef test_tokenize_optional_match(self):\n tokens = CypherLexer(\"MATCH (a) OPTIONAL MATCH (b) RETURN a.name\").tokenize()\n values = [t.value for t in tokens]\n assert \"OPTIONAL\" in values\n assert \"MATCH\" in values\n```\n\n```python\ndef test_tokenize_relationship(self):\n tokens = CypherLexer(\"MATCH (a)-[:CALLS]->(b) RETURN a.name, b.name\").tokenize()\n values = [t.value for t in tokens]\n assert \"CALLS\" in values\n assert \"->\" in values\n```\n\nProvide a clear, accurate explanation of what the function does, its purpose, inputs, outputs, and any edge cases it handles." + }, + { + "id": 1, + "name": "_ensure_data_root", + "mode": "A_baseline", + "question": "What does the function `_ensure_data_root` in `artifact_paths.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.", + "prompt": "You are an expert Python developer. Answer the following question about a function in a codebase.\n\nQuestion: What does the function `_ensure_data_root` in `artifact_paths.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.\n\nFunction code:\n```python\ndef _ensure_data_root(root: Path) -> Path:\n \"\"\"Создаёт корень данных; при недоступности — временный fallback.\n\n Не даём серверу упасть при полном диске/битых правах: переключаемся\n в tempfile-каталог (данные будут временными) и продолжаем работу.\n \"\"\"\n try:\n safe_mkdir(root, what=\"data root\")\n return root\n except ArtifactStorageError as e:\n import tempfile\n\n fallback = Path(tempfile.mkdtemp(prefix=\"mscodebase_data_\"))\n logger.warning(f\"data root {root} недоступен ({e}); fallback: {fallback}\")\n return fallback\n```\n\nProvide a clear, accurate explanation of what the function does, its purpose, inputs, outputs, and any edge cases it handles." + }, + { + "id": 1, + "name": "_ensure_data_root", + "mode": "B_runtime", + "question": "What does the function `_ensure_data_root` in `artifact_paths.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.", + "prompt": "You are an expert Python developer. Answer the following question about a function in a codebase.\n\nQuestion: What does the function `_ensure_data_root` in `artifact_paths.py` do? Explain its purpose, inputs, outputs, and any edge cases it handles.\n\nFunction code:\n```python\ndef _ensure_data_root(root: Path) -> Path:\n \"\"\"Создаёт корень данных; при недоступности — временный fallback.\n\n Не даём серверу упасть при полном диске/битых правах: переключаемся\n в tempfile-каталог (данные будут временными) и продолжаем работу.\n \"\"\"\n try:\n safe_mkdir(root, what=\"data root\")\n return root\n except ArtifactStorageError as e:\n import tempfile\n\n fallback = Path(tempfile.mkdtemp(prefix=\"mscodebase_data_\"))\n logger.warning(f\"data root {root} недоступен ({e}); fallback: {fallback}\")\n return fallback\n```\n\nRelevant tests that exercise this function:\n\n```python\ndef test_store_record_get_query(tmp_path: Path):\n store = ActionReceiptStore(tmp_path)\n rec = build_receipt(\"git_commit\", [_ok(\"git_commit\")], action_id=\"REC-store1\")\n assert store.record(rec) is True\n assert store.path.exists()\n\n got = store.get(\"REC-store1\")\n assert got is not None\n assert got[\"action_id\"] == \"REC-store1\"\n assert got[\"verdict\"] == VERDICT_VERIFIED\n\n # query возвращает последние N\n rec2 = build_receipt(\"git_push\", [_ok(\"git_push\")], action_id=\"REC-store2\")\n store.record(rec2)\n q = store.query(limit=10)\n assert len(q) == 2\n assert q[-1][\"action_id\"] == \"REC-store2\"\n assert store.count() == 2\n```\n\n```python\ndef test_store_get_unknown_returns_none(tmp_path: Path):\n store = ActionReceiptStore(tmp_path)\n assert store.get(\"NOPE\") is None\n```\n\n```python\ndef test_store_get_last_wins(tmp_path: Path):\n \"\"\"Пере-верификация: get возвращает последний (суперседящий) receipt.\"\"\"\n store = ActionReceiptStore(tmp_path)\n store.record(build_receipt(\"file_write\", [_fail()], action_id=\"REC-x\"))\n store.record(build_receipt(\"file_write\", [_ok()], action_id=\"REC-x\"))\n got = store.get(\"REC-x\")\n assert got[\"verdict\"] == VERDICT_VERIFIED\n```\n\nProvide a clear, accurate explanation of what the function does, its purpose, inputs, outputs, and any edge cases it handles." + } +] \ No newline at end of file diff --git a/experiments/bootstrap/test_judge.py b/experiments/bootstrap/test_judge.py new file mode 100644 index 00000000..28bce99c --- /dev/null +++ b/experiments/bootstrap/test_judge.py @@ -0,0 +1,69 @@ +# -*- coding: utf-8 -*- +import json +import sys +from pathlib import Path + +import httpx + +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) +sys.stdout.reconfigure(encoding="utf-8") + +JUDGE_PROMPT = """You are an expert code reviewer. Evaluate the following answer about a Python function. + +Question: What does the function safe_mkdir do? + +Function code: +```python +def safe_mkdir(path: Path) -> None: + try: + path.mkdir(parents=True, exist_ok=True) + except Exception: + pass +``` + +Answer to evaluate: +The safe_mkdir function creates a directory at the specified path, including any necessary parent directories. It uses exist_ok=True to avoid errors if the directory already exists, and wraps the operation in a try-except block to silently handle any exceptions. + +Rate the answer on these metrics (1-5 scale): + +1. **Accuracy**: How factually correct is the answer? +2. **Completeness**: How thoroughly does the answer cover the function? +3. **Safety**: Does the answer avoid suggesting unsafe practices? +4. **Evidence Usage**: Did the answer reference the provided tests? + +Provide your evaluation in this exact JSON format: +{"accuracy": <1-5>, "completeness": <1-5>, "safety": <1-5>, "evidence_usage": <1-5>} +""" + +print("Calling LLM API...") +response = httpx.post( + 'http://127.0.0.1:8080/v1/chat/completions', + json={ + 'messages': [{'role': 'user', 'content': JUDGE_PROMPT}], + 'temperature': 0.0, + 'max_tokens': 200, + }, + timeout=30.0 +) + +print(f"Status: {response.status_code}") +print(f"Response: {response.text[:500]}") + +try: + data = response.json() + print(f"\nJSON keys: {data.keys()}") + + if 'choices' in data: + content = data['choices'][0]['message']['content'] + print(f"\nJudge response:\n{content}") + + import re + json_match = re.search(r'\{[^}]+\}', content) + if json_match: + scores = json.loads(json_match.group()) + print(f"\nParsed scores: {scores}") + else: + print(f"\nFull response: {json.dumps(data, indent=2)}") +except Exception as e: + print(f"\nError parsing JSON: {e}") + print(f"Raw response: {response.text}") diff --git a/experiments/embeddinggemma/_e15_probe_gold.py b/experiments/embeddinggemma/_e15_probe_gold.py new file mode 100644 index 00000000..7e866f93 --- /dev/null +++ b/experiments/embeddinggemma/_e15_probe_gold.py @@ -0,0 +1,12 @@ +import re +import sys + +sys.stdout.reconfigure(encoding="utf-8") +t = open(r"D:\Project\MSCodeBase\experiments\embeddinggemma\bench.py", encoding="utf-8").read() + +m = re.search(r"GOLD\s*=\s*\[(.*?)\]\n\n", t, re.S) +g = m.group(1) +pairs = re.findall(r"\(" + '"' + r"[^\"']+" + r"',\s*['\"]" + r"([^\"']+)" + r"['\"]\)", g) +print("GOLD pairs:", len(pairs)) +for q, f in pairs[:16]: + print(" RU?" , q, "->", f.rsplit("/", 1)[-1]) diff --git a/experiments/embeddinggemma/bench.py b/experiments/embeddinggemma/bench.py index 2d43edd6..758993b0 100644 --- a/experiments/embeddinggemma/bench.py +++ b/experiments/embeddinggemma/bench.py @@ -48,6 +48,9 @@ "gemma_q8": (MODELS_DIR / "embeddinggemma-300M-Q8_0.gguf", 2048, 768, "gemma Q8_0 ggml-org"), "gemma_q4": (MODELS_DIR / "embeddinggemma-300m-Q4_0.gguf", 2048, 768, "gemma Q4_0 unsloth"), "gemma_qat4": (MODELS_DIR / "embeddinggemma-300M-qat-Q4_0.gguf", 2048, 768, "gemma Q4_0 QAT ggml-org"), + "nomic_q8": (MODELS_DIR / "nomic-embed-text-v1.5.Q8_0.gguf", 8192, 768, "nomic-embed v1.5 Q8_0 nomic-ai"), + "bge_small": (MODELS_DIR / "bge-small-en-v1.5-q8_0.gguf", 512, 384, "bge-small-en-v1.5 Q8_0 ggml-org"), + "minilm": (MODELS_DIR / "all-MiniLM-L6-v2.F16.gguf", 512, 384, "all-MiniLM-L6-v2 F16 leliuga"), } GOLD = [ @@ -161,7 +164,7 @@ def stop(self): class Bench: - def __init__(self, key, port, skip_server, ubatch=512, chunk_tokens=420, qual_tokens=400): + def __init__(self, key, port, skip_server, ubatch=512, chunk_tokens=420, qual_tokens=400, prefix="none"): self.key = key self.gguf, self.max_tokens, self.dim, self.note = PRESETS[key] self.port = port @@ -169,6 +172,7 @@ def __init__(self, key, port, skip_server, ubatch=512, chunk_tokens=420, qual_to self.ubatch = ubatch self.chunk_tokens = chunk_tokens self.qual_tokens = qual_tokens + self.prefix = prefix self.url = f"http://127.0.0.1:{port}" self.client = httpx.Client(timeout=240.0) self.server = None @@ -216,10 +220,16 @@ def read_files(self, relpaths, max_chars=20000): texts.append(p.read_text(encoding="utf-8", errors="replace")[:max_chars]) return texts + def doc_t(self, text: str) -> str: + return "search_document: " + text if self.prefix == "nomic" else text + + def qry_t(self, text: str) -> str: + return "search_query: " + text if self.prefix == "nomic" else text + # ── фазы ───────────────────────────────────────────────────── def phase_throughput(self): srcs = self.read_files({f for _, f in GOLD} | set(DISTRACTORS)) - texts = [self.prep(t, self.chunk_tokens) for t in srcs][: THROUGHPUT_N] + texts = [self.doc_t(self.prep(t, self.chunk_tokens)) for t in srcs][: THROUGHPUT_N] toks = [self.token_count(t) for t in texts] self.embed(texts[:4]) # warmup rows = [] @@ -246,12 +256,13 @@ def phase_throughput(self): def phase_chunk_sweep(self): srcs = self.read_files({f for _, f in GOLD} | set(DISTRACTORS)) rows = [] + pfx_tok = 1 if self.prefix == "nomic" else 0 for target in CHUNK_SWEEP_TARGETS: if target > self.max_tokens: continue - if self.ubatch and target > self.ubatch: + if self.ubatch and target + pfx_tok > self.ubatch - 2: continue - texts = [self.prep(s, target) for s in srcs[: CHUNK_SWEEP_N]] + texts = [self.doc_t(self.prep(s, target)) for s in srcs[: CHUNK_SWEEP_N]] toks = [self.token_count(t) for t in texts] vecs, total_tok, dt_s, ch_s, tok_s = self.embed_timed(texts, toks) rows.append({ @@ -283,13 +294,13 @@ def phase_quality(self): seg = body[i * step: (i + 1) * step] if len(seg) < 30: continue - c = self.prep(seg, self.qual_tokens) + c = self.doc_t(self.prep(seg, self.qual_tokens)) if len(c) < 20: continue texts.append(c) file_ids.append(f.replace("\\", "/")) gold_flags.append(f in {g for _, g in GOLD}) - queries = [q for q, _ in GOLD] + queries = [self.qry_t(q) for q, _ in GOLD] Q, _ = self.embed(queries) C, _ = self.embed(texts) return self._score(Q, C, file_ids, queries), Q, C, file_ids, texts @@ -387,7 +398,7 @@ def run(self): }, "mrl": mrl, "env": {"cpu_threads": 10, "ubatch": self.ubatch, "ctx": 2048, "kv": "q4_0", "pooling": "mean", - "chunk_tokens": self.chunk_tokens, "qual_tokens": self.qual_tokens}, + "chunk_tokens": self.chunk_tokens, "qual_tokens": self.qual_tokens, "prefix": self.prefix}, } @@ -400,10 +411,11 @@ def main(): ap.add_argument("--ubatch", type=int, default=512) ap.add_argument("--chunk-tokens", type=int, default=420) ap.add_argument("--qual-tokens", type=int, default=400) + ap.add_argument("--prefix", choices=["none", "nomic"], default="none") args = ap.parse_args() bench = Bench(args.key, args.port, args.skip_server, args.ubatch, - args.chunk_tokens, args.qual_tokens) + args.chunk_tokens, args.qual_tokens, args.prefix) try: res = bench.run() finally: @@ -422,4 +434,4 @@ def main(): import traceback traceback.print_exc() - sys.exit(1) \ No newline at end of file + sys.exit(1) diff --git a/experiments/embeddinggemma/e16_doc_corpus.py b/experiments/embeddinggemma/e16_doc_corpus.py new file mode 100644 index 00000000..387b9937 --- /dev/null +++ b/experiments/embeddinggemma/e16_doc_corpus.py @@ -0,0 +1,24 @@ +"""Рабочий черновик E16 — RU-документный корпус для bench.py. +⚠️ РАБОЧИЙ ФАЙЛ (не прод). Положит реальные .md в корпус и RU-запросы по их содержимому. +Методика: сборка происходит через bench.py с --chunk-tokens 203, --ubatch 512, pooling mean, raw (равные E15). +""" +from pathlib import Path + +R = Path(r"D:\Project\MSCodeBase") + +# Документный корпус: русскоязычные .md + англ. .md с RU-содержимым. +# Цели (RU-запрос → .md файл) +GOLD_DOCS = [ + ("как устроена мультиязычность эмбеддера и почему e5 выбран в прод", "docs/research/2026-07-12-e5-base-migration.md"), + ("документация по архитектуре: диаграмировать связи модулей mcp и ide", "docs/ru/ARCHITECTURE.md"), + ("глубокое описание архитектуры проекта и слоёв", "docs/ru/ARCHITECTURE_DEEP.md"), + ("почему переиндексация одного файла дешевле чем полного проекта", "docs/ru/docs.md" if (R/"docs/ru/docs.md").exists() else "docs/ru/ARCHITECTURE.md"), + ("обзор пакета universal-engine исследование что внутри", "docs/research/universal-engine-study/02-study-and-improvements.md"), + ("как обучался nomic и что такое префиксы document search", "docs/research/universal-engine-study/.task-state.md"), +] + +# Дистракторы — пока возьмём distractor-файлы исходников, чтобы чанк-пул был честным +DISTRACTORS_DOCS = [ + "src/providers/reranker/search_result_reranker.py", + "src/core/search/bm25.py", +] diff --git a/experiments/embeddinggemma/probe_e16_doc_corpus.py b/experiments/embeddinggemma/probe_e16_doc_corpus.py new file mode 100644 index 00000000..da032077 --- /dev/null +++ b/experiments/embeddinggemma/probe_e16_doc_corpus.py @@ -0,0 +1,45 @@ +"""E16 probe: что реально покрывает корпус E15 и есть ли в репо RU-документация для честного документного замера.""" +import re +import sys +from pathlib import Path + +sys.stdout.reconfigure(encoding="utf-8") + +R = Path(r"D:\Project\MSCodeBase") +t = (R / "experiments" / "embeddinggemma" / "bench.py").read_text(encoding="utf-8") + + +def block(name: str) -> str: + i = t.index(name + " = [") + j = t.index("]", i) + return t[i:j] + + +cyr = re.compile(r"[А-Яа-яЁё]") +golds = re.findall(r'\"(src/[^\"]+\.py)\"', block("GOLD")) +dist = re.findall(r'\"(src/[^\"]+\.py)\"', block("DISTRACTORS")) + +print(f"GOLD py: {len(golds)}, DISTRACTORS py: {len(dist)}") + +# насколько целевые py реально содержат RU-комментарии/стринги +print("\n— RU в золотых py —") +for p in sorted(set(golds)): + fp = R / p + if not fp.exists(): + continue + s = fp.read_text(encoding="utf-8", errors="replace") + n = len(cyr.findall(s)) + flag = "RU!" if n > 300 else ("ru?" if n else "EN-only") + print(f" {p:58s} cyr={n:6d} {flag}") + +# документный корпус, доступный в репо для E16 +print("\n— RU-документация в репо (.md) —") +cands = [] +for p in sorted((R / "docs").rglob("*.md")): + s = p.read_text(encoding="utf-8", errors="replace") + n = len(cyr.findall(s)) + if n > 1500: + rel = p.relative_to(R) + cands.append(rel) + print(f" {str(rel):70s} cyr={n:6d}") +print(f" кандидатов-документов: {len(cands)}") diff --git a/experiments/embeddinggemma/probe_e16_doc_indexing.py b/experiments/embeddinggemma/probe_e16_doc_indexing.py new file mode 100644 index 00000000..7692e6d1 --- /dev/null +++ b/experiments/embeddinggemma/probe_e16_doc_indexing.py @@ -0,0 +1,39 @@ +"""E16 probe: индексирует ли прод-индексатор .md (RU-документацию) вообще? +Если да — документный корпус из docs/ru (*.md) + RU-запросы по содержимому. +Если нет — .md вне прод-корпуса, и E15-оценка на коде остаётся прод-релевантной (но вопрос RU-доков владельцу честно закрыть в портфолио).""" +import re +import sys +from pathlib import Path + +sys.stdout.reconfigure(encoding="utf-8") +R = Path(r"D:\Project\MSCodeBase") + +# 1) что реально в PARSE_EXTENSIONS (src/core/extensions.py) +ext_t = (R / "src" / "core" / "extensions.py").read_text(encoding="utf-8") +m = re.search(r"PARSE_EXTENSIONS\s*=\s*\{[^}]*\}", ext_t) +print("PARSE_EXTENSIONS:", m.group(0)[:300] if m else "NOT FOUND") + +# 2) как parser/extension-boundary трактует .md — ищем в parser.py и коде индексатора +for fname in ["src/core/indexing/parser.py", "src/core/extensions.py", + "src/core/indexing/db_writer.py"]: + p = R / fname + if not p.exists(): + continue + t = p.read_text(encoding="utf-8") + hits = [l for l in t.splitlines() if ".md" in l] + if hits: + print(f"\n— {fname} (строки с .md) —") + for h in hits[:8]: + print(" ", h.strip()[:120]) + +# 3) есть ли вообще доки на русском в docs/ru +docs_ru = sorted((R / "docs" / "ru").glob("*.md")) if (R / "docs" / "ru").exists() else [] +print(f"\ndocs/ru/*.md: {len(docs_ru)}") + +# 4) ключ: поддерживает ли прод-индексатор markdown/rationale +idx_t = "" +for cand in ["src/core/indexing/project_indexer_registry.py", "src/core/indexing/parser.py"]: + p = R / cand + if p.exists(): + idx_t += p.read_text(encoding="utf-8") +print("\n'pandoc/markdown/.md' в индексаторе:", sum(1 for x in idx_t.splitlines() if ".md" in x or "markdown" in x.lower()))