From aedddf5ae4e0de717db838def12ac5a6cc591fcc Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 20:15:29 +0000 Subject: [PATCH 1/5] fix: repair the trust guarantees found by the 2026-09-26 deep review Every reproduced finding (F01-F16) gets a fix and a regression test: - verify: the code-state fingerprint is a canonical, length-delimited manifest bound to HEAD (renames, byte repartitions, empty files, modes, symlinks; unreadable files fail closed) (F01); nested workspace suites are planned, run and reported as coverage (F08); a runner found only as a dependency is inventory, not an extra obligation (F09); a tree that changes while the tests run is INCOMPLETE, never a signed PASS (F10); each run records a MAC'd verifier event (A01). - reuse: exact keys are lossless except whitespace (F04); near candidates must pass a semantic guard; artifacts are revalidated at serve time against their file digest and dependency contracts, and a missing atlas is "unknown", not "ok" (F05). - ledger: evidence is counted per event, so aliases of one git object vote once (F06); a rewritten lesson inherits no trust unless equivalent (F07); archive reasons are recorded and an idle/duplicate archive is not a refutation (F15); similar-but-opposite rules are never merged (F16). - context: the rendered block never exceeds the budget while claiming success, pointers are pending reads, spans/truncation are explicit (F02, F03, R13). - dash: every route checks Host; writes need a session token and the exact origin (F13); missing spend is unknown, not $0 (A10). - router: sparse cost fits use the mean log cost with an explicit variance prior (F11); infeasible budgets/targets are explicit (F12); outcomes are schema-validated, idempotent and labeled self-reported (A04). - bench: every labeled reuse tier is validated before timing; all sketch caches are cleared for cold rows (F14, A06). - imagine: an isolated checkout, not a sandbox; foreign runners are an explicit unsupported result (A05). - CI runs the Python prototype suites and the research recomputation (A02). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- .github/workflows/ci.yml | 37 +- .gitignore | 5 + bench/bench.mjs | 132 +++++-- global/guards/format-on-edit.sh | 12 +- landing/index.html | 2 +- reports/benchmarks.md | 99 +++--- src/cli.js | 102 +++++- src/commands.js | 12 +- src/context.js | 275 +++++++++++---- src/cortex.js | 40 ++- src/cost_report.js | 19 +- src/dash.html | 37 +- src/dash.js | 160 +++++++-- src/imagine.js | 74 ++-- src/learn_consolidate.js | 105 +++++- src/ledger.js | 68 +++- src/ledger_bridge.js | 65 +++- src/ledger_retention.js | 66 +++- src/ledger_store.js | 106 +++++- src/reuse.js | 309 +++++++++++++---- src/router/cost.js | 100 +++++- src/router/index.js | 164 +++++++-- src/router/policy.js | 27 +- src/router/registry.js | 40 ++- src/schema.js | 87 +++++ src/semantic_guard.js | 192 +++++++++++ src/stack.js | 144 ++++++-- src/substrate.js | 30 +- src/verify.js | 593 ++++++++++++++++++++++++++++---- test/bench.test.js | 29 ++ test/context.test.js | 82 ++++- test/cost_report.test.js | 4 + test/dash.test.js | 104 +++++- test/embed.test.js | 16 +- test/imagine.test.js | 16 + test/learn_consolidate.test.js | 115 ++++++- test/ledger_bridge.test.js | 60 +++- test/ledger_retention.test.js | 36 +- test/ledger_store.test.js | 92 +++++ test/reuse.test.js | 124 ++++++- test/router_math.test.js | 63 ++++ test/router_universal.test.js | 71 +++- test/schema.test.js | 67 ++++ test/semantic_guard.test.js | 70 ++++ test/stack.test.js | 28 +- test/verify.test.js | 310 ++++++++++++++++- 46 files changed, 3895 insertions(+), 494 deletions(-) create mode 100644 src/schema.js create mode 100644 src/semantic_guard.js create mode 100644 test/schema.test.js create mode 100644 test/semantic_guard.test.js diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c95f8faa..66e76869 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,6 +1,7 @@ # CI gate for every push and PR: the Node 20/22 test matrix (matches the ">=20" engines -# field; 18 is EOL) plus the shared quality gate (Biome, typecheck, ShellCheck, zero-dep -# assertion, version-drift, docs-drift, pack). The quality gate is the SAME reusable workflow +# field; 18 is EOL), the research contracts (Python prototype suites + recomputation), plus +# the shared quality gate (Biome, typecheck, ShellCheck, zero-dep assertion, version-drift, +# docs-drift, pack). The quality gate is the SAME reusable workflow # the version bump and release require, so none of the three can drift from the others. name: CI @@ -58,6 +59,38 @@ jobs: awk '/^not ok /{p=1} p{print} p&&/^ \.\.\.[[:space:]]*$/{p=0}' /tmp/win-test.log | head -200 exit "$ec" + # The research contracts (review A02): the two Python prototypes' own suites, the Theorem-D + # sanity checks (no data needed), and the recomputation of every corrected number from the + # replication package shipped in this repo. A research claim that stops recomputing fails CI + # like a broken unit test. + research: + name: Research (Python prototypes + recomputation) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-python@v6 + with: + python-version: "3.12" + - name: Install prototype test dependencies + run: | + python -m pip install --upgrade pip + python -m pip install \ + -r research/python-prototypes/impact_oracle/requirements.txt \ + -r research/python-prototypes/router_gate/requirements.txt + - name: impact_oracle tests + working-directory: research/python-prototypes/impact_oracle + run: python -m pytest -q + - name: router_gate tests + working-directory: research/python-prototypes/router_gate + run: python -m pytest -q + - name: Theorem sanity checks (no data) + run: python research/recompute_corrections.py --theorem-checks + - name: Recompute the corrections from the replication package + run: | + mkdir -p "$RUNNER_TEMP/rp" + tar -xzf research/empirical-refutation/replication_package.tar.gz -C "$RUNNER_TEMP/rp" + python research/recompute_corrections.py "$RUNNER_TEMP/rp/repro" + quality-gate: name: Quality gate uses: ./.github/workflows/reusable-quality-gate.yml diff --git a/.gitignore b/.gitignore index 118cb3a7..a7448fe1 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,11 @@ node_modules/ npm-debug.log* *.tgz +# Python bytecode / test caches (research prototypes) +__pycache__/ +*.pyc +.pytest_cache/ + # Forge runtime artifacts (generated, never committed) .forge/ *.log diff --git a/bench/bench.mjs b/bench/bench.mjs index 1e8b6930..3ba4d12c 100644 --- a/bench/bench.mjs +++ b/bench/bench.mjs @@ -166,8 +166,32 @@ function fillLedger(dir, { count, offset = 0, evidenceEvery = 4 }) { } } -/** n in-memory artifact claims, each with one test.run confirm so it clears SERVE_FLOOR. */ -function makeArtifacts(n) { +/** + * A real git object to cite as evidence: a throwaway one-commit repo, and its full commit id, + * checked to resolve (`git cat-file -e`) — the same resolution appendEvidence performs. The + * old fixture cited `bench:artifact:`, an untyped ref that val() caps below SERVE_FLOOR, so + * every "exact"/"near" row actually measured a MISS (review F14). + * @returns {{dir: string, oid: string}} + */ +export function resolvedEvidenceRepo() { + const dir = mkdtempSync(join(tmpdir(), "forge-bench-git-")); + const g = (...args) => execFileSync("git", args, { cwd: dir, stdio: "ignore" }); + g("init"); + g("config", "user.email", "bench@example.invalid"); + g("config", "user.name", "bench"); + g("config", "commit.gpgsign", "false"); + writeFileSync(join(dir, "fixture.txt"), "reuse bench fixture\n"); + g("add", "-A"); + g("commit", "-m", "bench fixture"); + const oid = execFileSync("git", ["rev-parse", "HEAD"], { cwd: dir, encoding: "utf8" }).trim(); + g("cat-file", "-e", oid); // throws if it does not resolve + return { dir, oid }; +} + +/** n in-memory artifact claims, each with one RESOLVED test.run confirm (a `git:` object id) + * so it genuinely clears SERVE_FLOOR. `ref` overrides the evidence ref — a stale/unresolved + * fixture for the validation self-test. */ +export function makeArtifacts(n, { ref }) { const rand = mulberry32(42); const claims = []; const specs = []; @@ -179,23 +203,43 @@ function makeArtifacts(n) { 0, ); if (!minted.ok) throw new Error(minted.reason); - const o = outcomeRecord({ - oracle: "test.run", - result: "confirm", - ref: `bench:artifact:${i}`, - author: "bench", - t: 0, - }); - minted.claim.evidence = o.ok ? [o.outcome] : []; + const o = outcomeRecord({ oracle: "test.run", result: "confirm", ref, author: "bench", t: 0 }); + if (!o.ok) throw new Error(`bench evidence rejected: ${o.reason}`); + minted.claim.evidence = [o.outcome]; claims.push(minted.claim); } return { claims, specs }; } -const coldSketches = (claims) => { - for (const c of claims) delete c._sketch; +/** Strip EVERY memoized sketch the lookup path caches — ledger.js's claim-text `_sketch`/ + * `_terms` and reuse.js's `_specSketch`/`_keySketch` — so a "cold" row behaves like a fresh + * process. (It used to delete `_sketch` only, which reuse never reads.) */ +export const coldSketches = (claims) => { + for (const c of claims) { + delete c._sketch; + delete c._terms; + delete c._specSketch; + delete c._keySketch; + } }; +/** + * A row is only labeled with a tier it actually exercised: run the lookup once, untimed, and + * abort the benchmark unless the expected tier comes back (review F14). + * @param {any[]} claims + * @param {string} spec + * @param {"exact"|"near"|"adapt"|"miss"} tier + */ +export function expectTier(claims, spec, tier) { + coldSketches(claims); + const r = lookup(claims, spec); + if (r.tier !== tier) + throw new Error( + `bench fixture invalid: expected a ${tier} lookup, got ${r.tier}${r.reasons?.length ? ` (${r.reasons.slice(0, 2).join("; ")})` : ""}`, + ); + return r; +} + // --------------------------------------------------------------------------- // The run // --------------------------------------------------------------------------- @@ -204,6 +248,17 @@ function environment() { let commit = "unknown"; try { commit = execFileSync("git", ["rev-parse", "HEAD"], { cwd: REPO_ROOT }).toString().trim(); + // Numbers measured on an uncommitted tree are not the numbers OF that commit — say so. + const dirty = execFileSync("git", ["status", "--porcelain"], { cwd: REPO_ROOT }) + .toString() + .trim(); + if (dirty) commit += " + uncommitted changes"; + } catch {} + // The filesystem the tmpdir fixtures live on changes the disk-bound rows (A06). + let fsType = "unknown"; + try { + if (platform() === "linux") + fsType = execFileSync("stat", ["-f", "-c", "%T", tmpdir()], { encoding: "utf8" }).trim(); } catch {} return { node: process.version, @@ -212,6 +267,7 @@ function environment() { memGB: Math.round(totalmem() / 2 ** 30), platform: platform(), arch: arch(), + fsType, commit, date: new Date().toISOString(), }; @@ -240,7 +296,7 @@ function runBenchmarks() { "atlas", "full build (this repo)", tFull, - `${atlas.files} files, ${atlas.symbols.length} symbols, ${atlas.edges.length} edges`, + `${atlas.files} files, ${atlas.symbols.length} symbols, ${atlas.edges.length} edges${atlas.capped ? `, CAPPED at ${atlas.cap} files` : `, cap ${atlas.cap ?? "none"} not reached`}`, ); const tIncr = timeIt( () => { @@ -335,10 +391,17 @@ function runBenchmarks() { tFp, fmtRate(fpSpecs.length / (tFp.median / 1000)), ); + const evidence = resolvedEvidenceRepo(); + cleanup.push(evidence.dir); for (const n of [100, 1000]) { - const { claims, specs } = makeArtifacts(n); + const { claims, specs } = makeArtifacts(n, { ref: `git:${evidence.oid}` }); const exactQ = specs[n >> 1]; const nearQ = `${specs[n >> 1]} gently`; // superset tokens → Jaccard ≈ 0.97 → near tier + const missQ = "configure the blue ocean lighthouse keeper rotation schedule"; + // Validate BEFORE timing: each row must exercise the tier it is labeled with. + expectTier(claims, exactQ, "exact"); + expectTier(claims, nearQ, "near"); + expectTier(claims, missQ, "miss"); let hit = null; const tExact = timeIt( () => { @@ -347,7 +410,14 @@ function runBenchmarks() { }, { runs: 10, warmup: 2 }, ); - push("reuse", `lookup exact @ ${n} artifacts`, tExact, `tier=${hit.tier}`); + push("reuse", `lookup exact hit, cold @ ${n} artifacts`, tExact, `tier=${hit.tier}`); + const tExactWarm = timeIt( + () => { + hit = lookup(claims, exactQ); // memoized sketches kept: a long-lived process + }, + { runs: 10, warmup: 2 }, + ); + push("reuse", `lookup exact hit, warm @ ${n} artifacts`, tExactWarm, `tier=${hit.tier}`); const tNear = timeIt( () => { coldSketches(claims); @@ -357,10 +427,18 @@ function runBenchmarks() { ); push( "reuse", - `lookup near (LSH) @ ${n} artifacts`, + `lookup near hit (LSH), cold @ ${n} artifacts`, tNear, `tier=${hit.tier}, j=${hit.jaccard?.toFixed(2) ?? "-"}`, ); + const tMiss = timeIt( + () => { + coldSketches(claims); + hit = lookup(claims, missQ); + }, + { runs: 5, warmup: 1 }, + ); + push("reuse", `lookup miss, cold @ ${n} artifacts`, tMiss, `tier=${hit.tier}`); } // --- context: assemble() on this repo for a representative task ---------------- @@ -409,17 +487,27 @@ const RESULT_HEADERS = ["suite", "benchmark", "median", "p95", "runs", "notes"]; const QUALITY_HEADERS = ["case (target)", "precision", "recall", "F1", "predicted", "truth"]; const SERIES_HEADERS = ["series", "precision", "recall", "F1", "ground truth"]; -/** Paper prototype vs this repo — two different methodologies, side by side and - * labeled, NEVER averaged or blended. The paper row is a constant from the - * whitepaper (Figure 5 / deliverable-package.md), not something this harness ran. */ +/** Paper prototype vs this repo — different methodologies, side by side and labeled, + * NEVER averaged or blended. The two paper rows are constants, not something this harness + * ran: the self-built demo (whitepaper Figure 5 / deliverable-package.md), which the + * pre-registered field study REFUTED, and that field study's own result on real co-change + * data (research/empirical-refutation; recomputed by the 2026-09-26 external review). The + * regex atlas measured below is a different (Node) graph — not the evaluated Python oracle. */ export function seriesRows(quality) { return [ [ - "paper prototype (Python, mutation-derived)", + "paper prototype, self-built demo (REFUTED)", "0.63", "1.00", "0.75", - "mutation testing against a real suite", + "mutation testing on the authors' own fixture", + ], + [ + "paper prototype, field study (pooled, 9 repos)", + "0.40", + "0.02", + "0.04", + "759 files' mined co-change (research/empirical-refutation)", ], [ "this repo (regex atlas, hand-labeled)", @@ -475,7 +563,7 @@ function resultsMarkdown(env, rows, quality) { "", `Edited-file-only baseline recall over the same cases: **${quality.baseline.recall.toFixed(2)}**.`, "", - "Two methodologies, side by side — different codebases, different ground-truth", + "Different methodologies, side by side — different codebases, different ground-truth", "derivations, so the rows are comparable in spirit only and are never blended:", "", formatTable(SERIES_HEADERS, seriesRows(quality), { markdown: true }), diff --git a/global/guards/format-on-edit.sh b/global/guards/format-on-edit.sh index 8523eb08..53b59cc9 100755 --- a/global/guards/format-on-edit.sh +++ b/global/guards/format-on-edit.sh @@ -17,9 +17,19 @@ fpath="$(forge_field file_path)" have() { command -v "$1" >/dev/null 2>&1; } run() { "$@" >/dev/null 2>&1 || true; } +# A project that formats with Biome must never be rewritten by a global prettier: the two +# disagree (line width, object wrapping), so every edit churned the whole file. Biome owns +# the files it is configured for; everything else (markdown included) is left alone. +biome_project() { [ -f "biome.json" ] || [ -f "biome.jsonc" ]; } + case "$fpath" in *.ts|*.tsx|*.js|*.jsx|*.mjs|*.cjs|*.json|*.css|*.scss|*.md|*.html|*.yaml|*.yml) - if have npx && [ -f "package.json" ]; then run npx --no-install prettier --write "$fpath"; fi + if biome_project; then + case "$fpath" in + *.md|*.html|*.yaml|*.yml) ;; + *) if have npx && [ -x "node_modules/.bin/biome" ]; then run npx --no-install biome format --write "$fpath"; fi ;; + esac + elif have npx && [ -f "package.json" ]; then run npx --no-install prettier --write "$fpath"; fi ;; *.py) if have ruff; then run ruff format "$fpath"; run ruff check --fix "$fpath"; diff --git a/landing/index.html b/landing/index.html index 4545b689..ba68be82 100644 --- a/landing/index.html +++ b/landing/index.html @@ -1783,7 +1783,7 @@ -
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory, foresight, and guardrails—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
886ms
End-to-end benchmark
Blast-radius scan
0.40ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
+
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory, foresight, and guardrails—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
880ms
End-to-end benchmark
Blast-radius scan
1.10ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
  /plugin install forgekit
 

Recommended · ambient guards on every prompt

diff --git a/reports/benchmarks.md b/reports/benchmarks.md index b52e2f9c..7bc8cd59 100644 --- a/reports/benchmarks.md +++ b/reports/benchmarks.md @@ -45,12 +45,22 @@ - **ledger / val()**: pure in-memory scoring; the fixture gives most claims 0–1 evidence records, and val() cost scales with evidence count — a heavily-evidenced ledger will be slower per claim. -- **reuse / lookup**: the memoized `_sketch` cache is stripped before every timed run, so - each run behaves like a fresh CLI process. The *exact* tier returns before any pool - sketching (normalized-string compare); the *near* tier pays MinHash-sketching the whole - candidate pool plus LSH banding — that difference is the point of reporting both. +- **reuse / lookup**: every row is VALIDATED before it is timed — the fixture's artifacts + cite a real, resolvable git object as their test evidence, and the harness aborts unless + the lookup returns the tier the row is labeled with (review F14, 2026-09-26: the previous + fixture cited untyped `bench:artifact:` refs, which val() caps below the serving floor, + so every "exact"/"near" row had actually measured a MISS; those older numbers are + invalid). "cold" rows strip every memoized sketch the lookup path caches (`_sketch`, + `_terms`, `_specSketch`, `_keySketch` — the old harness stripped only `_sketch`, which the + reuse ladder never reads), so each run behaves like a fresh CLI process; "warm" rows keep + them, like a long-lived process. The *exact* tier returns before any pool sketching + (identity-key compare); *near* and *miss* pay MinHash-sketching the candidate pool plus + LSH banding — that difference is the point of reporting them separately. - **context / assemble()**: warm atlas, empty ledger (the repo copy has no `.forge`), includes the real file reads for pinned items. Task: a three-symbol, one-file edit spec. + "complete"/"incomplete" is the assembler's own honest verdict: an item that could only + be delivered as a pointer or partial span is a pending read, so a large named file makes + this task's context incomplete at the default budget (review F02/F03). - **substrate / substrateCheck**: the whole deterministic gate — preflight grounding, routing rubric, up to 8 impact queries, reuse lookup, context assembly, scope decomposition, lessons, minimality, goal anchor — with `llm: false`. **No model latency @@ -90,9 +100,12 @@ now resolve to the exact symbol (`src/atlas.js:17 imports → src/util.js:conten What these numbers do **not** mean: n = 6 cases, one JavaScript repo, symbols chosen to be uniquely named (the atlas resolves ambiguous names to nothing — a separate, known -limitation). They are not comparable to the paper's numbers, which came from mutation -testing a Python codebase against a real test suite. The two appear side by side below, -labeled, and are never blended. +limitation). They are not comparable to the paper prototype's numbers: its 0.63 / 1.00 / +0.75 came from mutation testing on the authors' own fixture — a self-built demo that the +pre-registered field study REFUTED (pooled precision 0.40, recall 0.022, F1 0.042 over 759 +files' mined co-change in nine repositories; see `research/empirical-refutation/`). The +regex atlas here is a different, Node graph — not the evaluated Python oracle. All three +appear side by side below, labeled, and are never blended. > **History of this row.** The precision 0.90 / F1 0.92 this file carried until 2026-09-21 came > from a much smaller atlas (145 files) and a reverse walk that stopped at the direct @@ -114,57 +127,63 @@ labeled, and are never blended. ```json { - "node": "v24.19.0", - "cpu": "AMD EPYC Processor (with IBPB)", + "node": "v22.22.2", + "cpu": "Intel(R) Xeon(R) Processor @ 2.80GHz", "cores": 4, - "memGB": 8, - "platform": "win32", + "memGB": 16, + "platform": "linux", "arch": "x64", - "commit": "703da31d574c30d22bef019b1c8563ade0d0d6be", - "date": "2026-09-21T22:09:58.366Z" + "fsType": "ext2/ext3", + "commit": "d2abfa69fb77531199ffc67c5c076b524af69040 + uncommitted changes", + "date": "2026-09-26T20:14:48.438Z" } ``` ### Measured results -| suite | benchmark | median | p95 | runs | notes | -|-----------|---------------------------------------------|----------|---------|------|----------------------------------------| -| atlas | full build (this repo) | 530 ms | 622 ms | 5 | 455 files, 10498 symbols, 29728 edges | -| atlas | incremental rebuild (unchanged) | 339 ms | 359 ms | 5 | per-file hash cache hit | -| atlas | impact("claimText") (warm adjacency) | 0.40 ms | 1.08 ms | 30 | 51 files impacted | -| ledger | mint+put 1000 claims | 1854 ms | 1986 ms | 5 | 539/s | -| ledger | loadClaims at 1000 claims | 213 ms | 230 ms | 5 | full state from disk | -| ledger | mergeDirs 2×500-claim replicas (250 shared) | 4308 ms | 4409 ms | 3 | +250 claims, +313 records | -| ledger | val() over 1000 claims | 0.076 ms | 0.16 ms | 20 | 13,140,604/s (mean val 0.51) | -| reuse | fingerprint 2000 specs | 116 ms | 156 ms | 5 | 17,171/s | -| reuse | lookup exact @ 100 artifacts | 5.71 ms | 10.5 ms | 10 | tier=miss | -| reuse | lookup near (LSH) @ 100 artifacts | 4.76 ms | 5.18 ms | 5 | tier=miss, j=- | -| reuse | lookup exact @ 1000 artifacts | 52.8 ms | 88.5 ms | 10 | tier=miss | -| reuse | lookup near (LSH) @ 1000 artifacts | 46.7 ms | 89.8 ms | 5 | tier=miss, j=- | -| context | assemble() (this repo, 3-symbol task) | 12.6 ms | 30.0 ms | 10 | 4070/6000 tokens, 9 required, complete | -| substrate | substrateCheck (allowBuild, llm off) | 886 ms | 908 ms | 3 | 99 impacted files, route simple | +| suite | benchmark | median | p95 | runs | notes | +|-----------|----------------------------------------------|---------|---------|------|--------------------------------------------------------------| +| atlas | full build (this repo) | 794 ms | 970 ms | 5 | 496 files, 13278 symbols, 36899 edges, cap 20000 not reached | +| atlas | incremental rebuild (unchanged) | 385 ms | 550 ms | 5 | per-file hash cache hit | +| atlas | impact("claimText") (warm adjacency) | 1.10 ms | 1.56 ms | 30 | 57 files impacted | +| ledger | mint+put 1000 claims | 266 ms | 332 ms | 5 | 3,762/s | +| ledger | loadClaims at 1000 claims | 15.1 ms | 15.7 ms | 5 | full state from disk | +| ledger | mergeDirs 2×500-claim replicas (250 shared) | 229 ms | 241 ms | 3 | +250 claims, +313 records | +| ledger | val() over 1000 claims | 0.59 ms | 1.93 ms | 20 | 1,687,379/s (mean val 0.51) | +| reuse | fingerprint 2000 specs | 233 ms | 315 ms | 5 | 8,585/s | +| reuse | lookup exact hit, cold @ 100 artifacts | 1.15 ms | 10.6 ms | 10 | tier=exact | +| reuse | lookup exact hit, warm @ 100 artifacts | 0.51 ms | 4.98 ms | 10 | tier=exact | +| reuse | lookup near hit (LSH), cold @ 100 artifacts | 10.9 ms | 17.2 ms | 5 | tier=near, j=0.98 | +| reuse | lookup miss, cold @ 100 artifacts | 10.6 ms | 15.6 ms | 5 | tier=miss | +| reuse | lookup exact hit, cold @ 1000 artifacts | 3.74 ms | 5.02 ms | 10 | tier=exact | +| reuse | lookup exact hit, warm @ 1000 artifacts | 3.32 ms | 4.08 ms | 10 | tier=exact | +| reuse | lookup near hit (LSH), cold @ 1000 artifacts | 122 ms | 138 ms | 5 | tier=near, j=0.95 | +| reuse | lookup miss, cold @ 1000 artifacts | 131 ms | 141 ms | 5 | tier=miss | +| context | assemble() (this repo, 3-symbol task) | 17.4 ms | 19.7 ms | 10 | 4174/6000 tokens, 9 required, incomplete | +| substrate | substrateCheck (allowBuild, llm off) | 880 ms | 962 ms | 3 | 136 impacted files, route simple | ### Impact-oracle quality (hand-labeled cases, this repo) | case (target) | precision | recall | F1 | predicted | truth | |---------------|-----------|--------|------|-----------|-------| -| normalizeSpec | 0.12 | 1.00 | 0.21 | 17 | 2 | +| normalizeSpec | 0.11 | 1.00 | 0.20 | 18 | 2 | | evalImpact | 0.29 | 1.00 | 0.44 | 7 | 2 | -| isStale | 0.20 | 1.00 | 0.33 | 30 | 6 | -| mergeStates | 0.16 | 1.00 | 0.28 | 25 | 4 | -| claimText | 0.16 | 1.00 | 0.27 | 51 | 8 | -| contentHash | 0.11 | 1.00 | 0.21 | 87 | 10 | +| isStale | 0.19 | 1.00 | 0.32 | 37 | 7 | +| mergeStates | 0.15 | 1.00 | 0.27 | 26 | 4 | +| claimText | 0.18 | 1.00 | 0.30 | 57 | 10 | +| contentHash | 0.11 | 1.00 | 0.20 | 100 | 11 | | mean of 6 | 0.17 | 1.00 | 0.29 | | | -Edited-file-only baseline recall over the same cases: **0.27**. +Edited-file-only baseline recall over the same cases: **0.26**. -Two methodologies, side by side — different codebases, different ground-truth +Different methodologies, side by side — different codebases, different ground-truth derivations, so the rows are comparable in spirit only and are never blended: -| series | precision | recall | F1 | ground truth | -|--------------------------------------------|-----------|--------|------|-----------------------------------------------| -| paper prototype (Python, mutation-derived) | 0.63 | 1.00 | 0.75 | mutation testing against a real suite | -| this repo (regex atlas, hand-labeled) | 0.17 | 1.00 | 0.29 | 6 hand-labeled cases (bench/impact_cases.mjs) | +| series | precision | recall | F1 | ground truth | +|------------------------------------------------|-----------|--------|------|------------------------------------------------------------| +| paper prototype, self-built demo (REFUTED) | 0.63 | 1.00 | 0.75 | mutation testing on the authors' own fixture | +| paper prototype, field study (pooled, 9 repos) | 0.40 | 0.02 | 0.04 | 759 files' mined co-change (research/empirical-refutation) | +| this repo (regex atlas, hand-labeled) | 0.17 | 1.00 | 0.29 | 6 hand-labeled cases (bench/impact_cases.mjs) | diff --git a/src/cli.js b/src/cli.js index 555427f2..c50fb7b8 100755 --- a/src/cli.js +++ b/src/cli.js @@ -350,6 +350,11 @@ HANDLERS.stack = async (argv) => { row("frameworks", s.frameworks); row("pkg mgrs", s.packageManagers); row("test", s.testCommands); + // Runners that are only installed (a devDependency) are inventory, never a required suite. + row( + "available", + (s.testInventory ?? []).filter((c) => !s.testCommands.includes(c)), + ); row("tools", s.tools); row("notes", s.notes); console.log(` ${"evidence:".padEnd(11)} ${s.evidence.join(", ")}`); @@ -758,13 +763,16 @@ HANDLERS.ledger = async (argv) => { // --fix re-addresses claims still stored under their pre-CRLF-fold id. Reads accept // that address either way, so this is not a repair — it is what stops one fact living // at two addresses once a teammate on another platform mints its current form. - const migration = args.includes("--fix") ? ls.migrateAddresses(dir) : null; + // `--fix --dry-run` previews the migration without writing (A11). + const migration = args.includes("--fix") + ? ls.migrateAddresses(dir, { dryRun: argv.includes("--dry-run") }) + : null; const r = ls.verify(dir); if (json) return console.log(JSON.stringify(migration ? { ...r, migration } : r, null, 2)); if (migration) { const { migrated, merged, failed } = migration; console.log( - ` migrated ${migrated.length} claim(s) to their current address, merged ${merged.length} into an existing twin${failed.length ? `, ${failed.length} failed` : ""}`, + ` ${migration.dryRun ? "(dry run) would migrate" : "migrated"} ${migrated.length} claim(s) to their current address, ${migration.dryRun ? "would merge" : "merged"} ${merged.length} into an existing twin${failed.length ? `, ${failed.length} failed` : ""}`, ); } console.log(` ${r.ok ? "OK" : "ISSUES"} — ${r.claims} claim(s), ${r.outcomes} outcome(s)`); @@ -931,6 +939,16 @@ HANDLERS.ledger = async (argv) => { ]; for (const a of r.archive.slice(0, 20)) lines.push(` ${a.id.slice(0, 12)} ${a.reason}`); if (r.archive.length > 20) lines.push(` … ${r.archive.length - 20} more (--json for all)`); + // Similar-but-opposite pairs are never archived as duplicates (review F16): a person decides. + const conflicts = d?.conflicts ?? []; + if (conflicts.length) { + lines.push( + "", + ` kept apart — similar but conflicting (review, then retract one): ${conflicts.length}`, + ); + for (const c of conflicts.slice(0, 10)) + lines.push(` ${c.a.slice(0, 12)} ↔ ${c.b.slice(0, 12)} ${c.conflicts}`); + } lines.push( "", dryRun @@ -1180,6 +1198,8 @@ HANDLERS.reuse = async (argv) => { jaccard: r.jaccard, similarity: r.similarity, sim: r.sim, + revalidation: r.revalidation?.status, + requiresRevalidation: r.requiresRevalidation === true, reasons: r.reasons, }, null, @@ -1197,8 +1217,14 @@ HANDLERS.reuse = async (argv) => { console.log( ` claim ${a.id.slice(0, 12)} — \`forge ledger blame ${a.id.slice(0, 8)}\` for its proof`, ); + if (r.tier === "near") + console.log(" near tier: a reworded match — review the diff before reusing it as-is"); if (r.tier === "adapt") console.log(" adapt tier: inject as a verified starting point, generate only the delta"); + if (r.requiresRevalidation) + console.log( + ` NOT revalidated: ${(r.revalidation?.unknown ?? []).join(", ")} — check before use`, + ); } for (const why of r.reasons) console.log(` note: ${why}`); return; @@ -1215,7 +1241,9 @@ HANDLERS.reuse = async (argv) => { return; } const { repoLedger } = await import("./ledger_store.js"); - const desc = ru.describeFile(root, file); + // With an atlas, each dependency's declaration is fingerprinted, so a later signature + // change invalidates the artifact (review F05). + const desc = ru.describeFile(root, file, { atlas: loadAtlas(root) }); const r = ru.mintArtifact( repoLedger(root), { spec, form: "module", ...desc }, @@ -1282,11 +1310,14 @@ HANDLERS.context = async (argv) => { nowDay: epochDay(), ...(budget ? { budget } : {}), }); + // --block delivers the assembled context itself — the spans the summary talks about (R13). + const withBlock = argv.includes("--block"); if (json) { const { block, ...rest } = r; - return console.log(JSON.stringify(rest, null, 2)); - } - console.log(renderContext(r)); + console.log(JSON.stringify(withBlock ? r : rest, null, 2)); + } else if (withBlock) { + console.log(r.block); + } else console.log(renderContext(r)); if (!r.ok) process.exitCode = 1; return; }; @@ -1512,12 +1543,22 @@ HANDLERS.verify = async (argv) => { ); if (t.notExecuted?.length) console.log(` suites skipped: ${t.notExecuted.join(", ")} (no built-in executor)`); + // Coverage (review F08): which package dirs the verdict actually speaks for. + const cov = t.coverage; + if (cov && cov.required.length > 1) + console.log( + ` packages: ${cov.covered.length}/${cov.required.length} covered${ + cov.uncovered.length ? ` — no verdict for ${cov.uncovered.join(", ")}` : "" + }${cov.excluded.length ? ` (${cov.excluded.length} excluded)` : ""}`, + ); + if (t.mutated) + console.log(" ! the code changed while the tests ran — the verdict is not bound to it"); console.log(` symbols checked: ${r.provenance.symbolsChecked}`); if (r.unknown.length) console.log( ` ! not in codebase (possible hallucination): ${r.unknown.slice(0, 12).join(", ")}`, ); - console.log(` provenance: .forge/provenance.json`); + console.log(` provenance: .forge/provenance.json (run ${r.provenance.event?.runId})`); // BLOCKED is reserved for a runner that actually FAILED; anything that never ran // to completion is NOT VERIFIED (still exit 1 — unverified is not a pass). const verdict = r.ok @@ -2204,7 +2245,15 @@ async function routeUniversalCli(argv) { const { loadRegistry } = await import("./router/registry.js"); const json = argv.includes("--json"); const val = (flag) => (argv.includes(flag) ? argv[argv.indexOf(flag) + 1] : undefined); - const VALUED = new Set(["--objective", "--provider", "--model", "--cost", "--depth"]); + const VALUED = new Set([ + "--objective", + "--provider", + "--model", + "--cost", + "--depth", + "--attempt", + "--verify-run", + ]); const words = argv .slice(1) .filter((a, i, arr) => !a.startsWith("--") && !VALUED.has(arr[i - 1] ?? "")); @@ -2246,7 +2295,7 @@ async function routeUniversalCli(argv) { if (!task) { console.error( 'usage: forge route universal "" [--objective match-best-single|target:

|value:<$>|budget:<$>] [--provider |any] [--depth ] [--json]\n' + - ' forge route outcome "" --model --pass|--fail [--cost ]\n' + + ' forge route outcome "" --model --pass|--fail [--cost ] [--attempt ] [--verify-run ]\n' + " forge route fit | forge route models", ); process.exitCode = 1; @@ -2256,10 +2305,19 @@ async function routeUniversalCli(argv) { const passed = argv.includes("--pass") ? true : argv.includes("--fail") ? false : undefined; const cost = val("--cost") !== undefined ? Number(val("--cost")) : null; try { - const row = U.recordOutcome(root, { task, model: val("--model"), passed, cost }); + const row = U.recordOutcome(root, { + task, + model: val("--model"), + passed, + cost, + attemptId: val("--attempt") ?? null, + verifyRunId: val("--verify-run") ?? null, + }); if (json) return console.log(JSON.stringify(row, null, 2)); console.log( - ` recorded ${row.model} ${row.passed ? "pass" : "fail"} for task ${row.task} (.forge/route_outcomes.jsonl)`, + row.duplicate + ? ` attempt ${row.attemptId} was already recorded — not counted twice` + : ` recorded ${row.model} ${row.passed ? "pass" : "fail"} (${row.provenance}) for task ${row.task} (.forge/route_outcomes.jsonl)`, ); } catch (e) { console.error(` ${e.message}`); @@ -2279,9 +2337,19 @@ async function routeUniversalCli(argv) { process.exitCode = 1; return; } - if (json) return console.log(JSON.stringify(rec, null, 2)); + if (json) { + console.log(JSON.stringify(rec, null, 2)); + if (!rec.ok) process.exitCode = 1; + return; + } if (!rec.ok) { - console.error(` ${rec.reason}`); + console.error(` ${rec.feasible === false ? "INFEASIBLE — " : ""}${rec.reason}`); + // F12: the least-bad cascade is shown only as an explicit, labeled fallback. + const fb = rec.fallback; + if (fb) + console.error( + ` fallback (does NOT meet the objective): ${fb.cascade.map((c) => c.model).join(" → ")} · P(success) ${fb.pSuccess.toFixed(2)} · expected $${fb.expectedCost.toFixed(3)} (up to $${fb.maxPossibleCost.toFixed(3)} if every attempt runs)`, + ); process.exitCode = 1; return; } @@ -2294,7 +2362,7 @@ async function routeUniversalCli(argv) { ); }); console.log( - `\n P(success) ${rec.pSuccess.toFixed(2)} · expected cost $${rec.expectedCost.toFixed(3)} · best single: ${rec.bestSingle.model} ${rec.bestSingle.pSuccess.toFixed(2)} at $${rec.bestSingle.expectedCost.toFixed(3)}`, + `\n P(success) ${rec.pSuccess.toFixed(2)} · expected cost $${rec.expectedCost.toFixed(3)} (not a cap; up to $${rec.maxPossibleCost.toFixed(3)} if every attempt runs) · best single: ${rec.bestSingle.model} ${rec.bestSingle.pSuccess.toFixed(2)} at $${rec.bestSingle.expectedCost.toFixed(3)}`, ); console.log( ` ${rec.candidates} candidate model(s), ${rec.cascadesEvaluated} cascade(s) compared · fit: ${rec.fit.origin}`, @@ -2532,7 +2600,7 @@ HANDLERS.imagine = async (argv) => { } catch {} // not a repo / no git → dryRun reports its own precondition failure if (dirty) { console.error( - "\n imagine --run refused: the working tree is dirty and the sandbox runs HEAD,\n" + + "\n imagine --run refused: the working tree is dirty and the isolated checkout runs HEAD,\n" + " so your uncommitted changes would NOT be in the dry-run. Commit or stash them,\n" + " or pass --allow-dirty to knowingly measure the last commit instead.", ); @@ -2564,7 +2632,9 @@ HANDLERS.imagine = async (argv) => { process.exitCode = 1; return; } - console.log(`\n dry-run (sandboxed worktree of HEAD · ${d.runner}):`); + console.log( + `\n dry-run (isolated checkout of HEAD — not a security sandbox · ${d.runner ?? "node --test"}):`, + ); console.log( ` pass ${d.passed} · fail ${d.failed} · ${d.durationMs}ms · worktree ${d.worktree}`, ); diff --git a/src/commands.js b/src/commands.js index 8bce40f4..4d77ea7d 100644 --- a/src/commands.js +++ b/src/commands.js @@ -106,7 +106,17 @@ export const COMMANDS = { ledger: "evidence-referenced memory — stats / verify / show / blame / query / compact / at / diff / root / ratify / retract / merge / sync / import", reuse: "proof-carrying code cache — query / mint --file / stats", - context: "budgeted context assembly + completeness gate — what an edit NEEDS known", + context: { + summary: + "budgeted context assembly + completeness gate — what an edit NEEDS known, delivered or owed", + usage: 'forge context "" [--budget ] [--block] [--json]', + flags: [ + { flag: "--budget ", desc: "assembly budget (chars/3.6 estimate; default 6000)" }, + { flag: "--block", desc: "print the assembled context block itself (what gets delivered)" }, + { flag: "--json", desc: "machine-readable result (add --block to include the block)" }, + ], + examples: ['forge context "update computeTax in src/tax.js" --block'], + }, preflight: "assumption check — what a task names that the repo doesn't define", config: "provider setup — show / switch / add providers, set default model", route: diff --git a/src/context.js b/src/context.js index 2feac0f8..3aec0849 100644 --- a/src/context.js +++ b/src/context.js @@ -2,11 +2,13 @@ // gate (docs/plans/substrate-v2/04-context-assembly.md). Two failures die here: // over-stuffing (everything competes for the window on equal terms — P3 of the // paper) and under-supplying (the agent edits a symbol without its callers, tests, -// or the team's lessons, then "assumes"). Selection gets an objective function -// (greedy knapsack by value density, with a compression ladder instead of silent -// drops) and sufficiency becomes a COMPUTED SET: required knowledge R(edit) from -// the atlas, missing = R \ covered — auto-fetched when resolvable, asked as a -// derived M2 question when not. Context insufficiency stops being a feeling. +// or the team's lessons, then "assumes"). Selection gets an objective (a greedy +// value-density heuristic with a compression ladder instead of silent drops — no +// approximation guarantee is claimed) and sufficiency becomes a COMPUTED SET: required +// knowledge R(edit) from the atlas, missing = R \ covered — auto-fetched when resolvable, +// asked as a derived M2 question when not. "Covered" means DELIVERED in the block: a +// pointer to a file is a pending read, not coverage (review F03), and a block that cannot +// fit the budget says so instead of claiming completion (review F02). import { existsSync, readFileSync } from "node:fs"; import { basename, dirname, join } from "node:path"; import { has as atlasHas, query as atlasQuery, impact } from "./atlas.js"; @@ -14,8 +16,20 @@ import { claimText, val } from "./ledger.js"; import { loadClaims, repoLedger } from "./ledger_store.js"; import { referencedEntities } from "./preflight.js"; -/** chars → tokens heuristic (calibrated in P8; consistent with the reuse estimator). */ +/** chars → tokens ESTIMATE (chars/3.6, consistent with the reuse estimator). It is not a + * model tokenizer: a hard window limit needs the target model's own tokenizer, so treat + * every budget here as an estimate with that stated basis. */ export const tokensOf = (text) => Math.ceil(String(text).length / 3.6); +/** How `tokens` is measured — reported with every assembly so nobody mistakes it for exact. */ +export const TOKEN_ESTIMATE = "chars/3.6 estimate of the rendered block"; +/** Items are joined with this separator; its cost is budgeted like any other text. */ +const SEP = "\n\n"; +/** Direct dependents listed by name before the rest are summarized as omitted. */ +const DEPS_SHOWN = 12; +/** Lines of source shown after a definition's line in a symbol-span variant. */ +const SPAN_AFTER = 40; +const SPAN_BEFORE = 2; +const HEAD_LINES = 25; /** Lessons must be THIS trusted to enter the required set (spec §3: lessons*(S)). */ export const LESSON_REQUIRED_VAL = 0.8; @@ -100,23 +114,93 @@ export function requiredSet(root, task, { atlas = null, claims = [], nowDay = 0 // An item is one injectable unit with a COMPRESSION LADDER: granularity variants from // full text down to a one-line pointer. The optimizer may downgrade an item instead of -// dropping it — compression is a lossy move with a known cost, chosen explicitly, -// never by scroll-off (spec §2). -function fileItem(root, rel, { covers, source, score }) { +// dropping it — compression is a lossy move with a known cost, chosen explicitly, never by +// scroll-off (spec §2). EVERY VARIANT CARRIES ITS OWN COVERAGE (review F03): the full file +// covers all its keys; a symbol span covers the definitions whose declaration line it shows; +// the first-25-lines head covers only what lies inside it; a pointer (`- read `) +// covers NOTHING — it creates a pending read obligation. Availability is not delivery. + +/** One variant: its rendered text, estimated tokens, the keys it satisfies, the keys it only + * points at (pending reads), and what it truncated. */ +const variant = (gran, text, covers, pending = [], truncated = null) => ({ + gran, + text, + tokens: tokensOf(text), + covers, + pending, + ...(truncated ? { truncated } : {}), +}); + +/** + * All variants of one file, given the required keys it serves: `needs` entries are + * {key, kind: "def"|"file"|"tests", line?}. Ordered largest → smallest; a variant that is not + * smaller than the previous one is skipped. + */ +function fileItem(root, rel, { needs, source, score }) { const text = readRel(root, rel); if (text === null) return null; - const head = text.split("\n").slice(0, 25).join("\n"); - const variants = [ - { gran: "full", text: `// ${rel}\n${text}`, tokens: tokensOf(text) }, - { gran: "head", text: `// ${rel} (first 25 lines)\n${head}`, tokens: tokensOf(head) }, - { gran: "pointer", text: `- read ${rel}`, tokens: 8 }, - ]; - return { id: `${source}:${rel}`, source, covers, score, variants }; + const lines = text.split("\n"); + const total = lines.length; + const keys = needs.map((n) => n.key); + const defs = needs.filter((n) => n.kind === "def" && Number.isFinite(n.line)); + const variants = [variant("full", `// ${rel}\n${text}`, keys)]; + // Symbol span: the lines around the requested definitions, when the atlas knows them. + if (defs.length) { + const from = Math.max(1, Math.min(...defs.map((d) => d.line)) - SPAN_BEFORE); + const to = Math.min(total, Math.max(...defs.map((d) => d.line)) + SPAN_AFTER); + if (from > 1 || to < total) { + const covers = defs.filter((d) => d.line >= from && d.line <= to).map((d) => d.key); + variants.push( + variant( + "span", + `// ${rel}:${from}-${to} of ${total} (definition span)\n${lines.slice(from - 1, to).join("\n")}`, + covers, + keys.filter((k) => !covers.includes(k)), + { shownLines: [from, to], totalLines: total }, + ), + ); + } + } + if (total > HEAD_LINES) { + // The head covers a definition only if its declaration line is inside the head; a whole + // file or test file is never "covered" by its first 25 lines. + const covers = needs + .filter((n) => n.kind === "def" && Number.isFinite(n.line) && n.line <= HEAD_LINES) + .map((n) => n.key); + variants.push( + variant( + "head", + `// ${rel} (first ${HEAD_LINES} of ${total} lines)\n${lines.slice(0, HEAD_LINES).join("\n")}`, + covers, + keys.filter((k) => !covers.includes(k)), + { shownLines: [1, HEAD_LINES], totalLines: total }, + ), + ); + } + variants.push(variant("pointer", `- read ${rel}`, [], keys)); + const ladder = []; + for (const v of variants.sort((a, b) => b.tokens - a.tokens)) + if (!ladder.length || v.tokens < ladder[ladder.length - 1].tokens) ladder.push(v); + return { id: `${source}:${rel}`, source, covers: keys, score, variants: ladder }; } /** * Assemble the context for a task: pinned required items (downgraded before dropped), * optional items greedily by value density, and the missing set as derived questions. + * + * Honesty contract (review F02/F03): + * - `tokens` is measured on the RENDERED block (labels and separators included) with the + * chars/3.6 estimate (`tokenEstimate` says so). The block never exceeds `budget` by that + * measure: when even pointers cannot fit, required items are DROPPED (lowest score first) + * and reported, with `overflow: true`. + * - `covered` holds only keys whose content was actually delivered; `pending` holds keys the + * block merely points at (a pointer, or a partial span/head) — read obligations; `missing` + * holds keys neither delivered nor pointed at (unresolvable, or dropped on overflow). + * - `ok` means every required key was DELIVERED within budget: no missing, no pending, no + * overflow. It is syntactic delivery, not semantic sufficiency. + * - Optional items are chosen greedily by value density (score per token) with per-source + * diminishing returns — a heuristic, with no knapsack or set-cover guarantee (the + * per-source discount breaks the preconditions those guarantees need). * @param {string} root * @param {string} task * @param {{budget?:number, atlas?:any, claims?:any[], nowDay?:number}} [opts] @@ -131,29 +215,46 @@ export function assemble( const required = requiredSet(root, task, { atlas, claims: allClaims, nowDay }); // --- build candidate items, keyed by what they cover ------------------------------- + // File-backed keys are grouped per file first, so one file is one item whose variants + // know exactly which of its keys each one delivers. + /** @type {Map} */ + const files = new Map(); + const need = (rel, n, source, score) => { + const f = files.get(rel) ?? { needs: [], source, score }; + f.needs.push(n); + if (score > f.score) { + f.score = score; + f.source = source; + } + files.set(rel, f); + }; const items = []; + /** @type {{id:string, shown:number, total:number, omitted:string[]}[]} */ + const truncated = []; for (const r of required) { if (!r.resolvable) continue; if (r.kind === "def") { const hit = atlasQuery(atlas, r.name).find((s) => s.name === r.name || s.qname === r.name); - if (hit?.file) { - const it = fileItem(root, hit.file, { covers: [r.key], source: "def", score: 1 }); - if (it) items.push(it); - } + if (hit?.file) need(hit.file, { key: r.key, kind: "def", line: hit.line }, "def", 1); } else if (r.kind === "file") { - const it = fileItem(root, r.name, { covers: [r.key], source: "def", score: 1 }); - if (it) items.push(it); + need(r.name, { key: r.key, kind: "file" }, "def", 1); } else if (r.kind === "tests") { - const it = fileItem(root, r.name, { covers: [r.key], source: "tests", score: 0.9 }); - if (it) items.push(it); + need(r.name, { key: r.key, kind: "tests" }, "tests", 0.9); } else if (r.kind === "deps" && atlas) { - const hop1 = impact(atlas, r.name, { maxHops: 1 }) - .impacted.filter((x) => x.hopDistance === 1) - .slice(0, 12); + const hop1 = impact(atlas, r.name, { maxHops: 1 }).impacted.filter( + (x) => x.hopDistance === 1, + ); + const shown = hop1.slice(0, DEPS_SHOWN); + const omitted = hop1.slice(DEPS_SHOWN).map((x) => `${x.node.name} (${x.node.file})`); + if (omitted.length) + truncated.push({ id: `deps:${r.name}`, shown: shown.length, total: hop1.length, omitted }); const text = hop1.length ? [ `direct dependents of ${r.name} (edit these with it or verify them):`, - ...hop1.map((x) => ` - ${x.node.name} (${x.node.file}, via ${x.edgeKinds[0]})`), + ...shown.map((x) => ` - ${x.node.name} (${x.node.file}, via ${x.edgeKinds[0]})`), + ...(omitted.length + ? [` … and ${omitted.length} more, omitted here (\`forge impact ${r.name}\`)`] + : []), ].join("\n") : `no direct dependents of ${r.name} found in the atlas`; items.push({ @@ -161,22 +262,33 @@ export function assemble( source: "deps", covers: [r.key], score: 1, - variants: [{ gran: "full", text, tokens: tokensOf(text) }], + variants: [ + variant("full", text, [r.key]), + variant("pointer", `- dependents: \`forge impact ${r.name}\``, [], [r.key]), + ], }); } else if (r.kind === "lesson") { const c = allClaims.find((x) => x.id === r.name); if (c) { const text = `lesson (val ${val(c, nowDay).toFixed(2)}): ${c.body.correctedBehavior}`; + const id8 = c.id.slice(0, 8); items.push({ - id: `lesson:${c.id.slice(0, 8)}`, + id: `lesson:${id8}`, source: "lesson", covers: [r.key], score: 0.95, - variants: [{ gran: "full", text, tokens: tokensOf(text) }], + variants: [ + variant("full", text, [r.key]), + variant("pointer", `- lesson: \`forge ledger show ${id8}\``, [], [r.key]), + ], }); } } } + for (const [rel, f] of files) { + const it = fileItem(root, rel, f); + if (it) items.push(it); + } // Optional extras: trusted scope-matching facts (nice-to-have, never required). for (const c of allClaims) { if (c.kind !== "fact" || c.tombstone) continue; @@ -188,37 +300,51 @@ export function assemble( source: "fact", covers: [], score: 0.3 + 0.4 * v, - variants: [{ gran: "full", text, tokens: tokensOf(text) }], + variants: [variant("full", text, [])], }); } - // A symbol's definition and an explicitly named file often resolve to the SAME file — - // merge items by id (union of covered keys) so one span never gets injected twice. - const byId = new Map(); - for (const it of items) { - const prev = byId.get(it.id); - if (prev) { - prev.covers = [...new Set([...prev.covers, ...it.covers])]; - prev.score = Math.max(prev.score, it.score); - } else byId.set(it.id, it); - } - const merged = [...byId.values()]; - // --- selection: pin required coverage, downgrade before dropping ------------------- - const pinned = merged.filter((i) => i.covers.length); - const optional = merged + const pinned = items.filter((i) => i.covers.length).sort((a, b) => (a.id < b.id ? -1 : 1)); + // Value density: score per token of the item's full variant (ties by id, deterministic). + const density = (i) => i.score / Math.max(1, i.variants[0].tokens); + const optional = items .filter((i) => !i.covers.length) - .sort((a, b) => b.score - a.score || (a.id < b.id ? -1 : 1)); - const chosen = pinned.map((i) => ({ item: i, v: 0 })); // v = variant index - const used = () => chosen.reduce((n, c) => n + c.item.variants[c.v].tokens, 0); + .sort((a, b) => density(b) - density(a) || (a.id < b.id ? -1 : 1)); + let chosen = pinned.map((i) => ({ item: i, v: 0 })); // v = variant index + const sepTokens = tokensOf(SEP); + // Per-piece ceilings plus separators bound the rendered block's estimate from above. + const used = () => + chosen.reduce((n, c) => n + c.item.variants[c.v].tokens, 0) + + Math.max(0, chosen.length - 1) * sepTokens; // Downgrade the largest pinned item one rung at a time until the pins fit the budget. while (used() > budget) { const cand = chosen .filter((c) => c.v < c.item.variants.length - 1) - .sort((a, b) => b.item.variants[b.v].tokens - a.item.variants[a.v].tokens)[0]; - if (!cand) break; // everything is already a pointer — required coverage beats budget + .sort( + (a, b) => + b.item.variants[b.v].tokens - a.item.variants[a.v].tokens || + (a.item.id < b.item.id ? -1 : 1), + )[0]; + if (!cand) break; cand.v++; } + // Everything is already at its smallest rung and still does not fit: the budget cannot + // hold the required set. Drop pinned items (lowest score first, then largest) and SAY so — + // never return a block over budget, and never call it complete (F02). + /** @type {string[]} */ + const dropped = []; + while (used() > budget && chosen.length) { + const worst = [...chosen].sort( + (a, b) => + a.item.score - b.item.score || + b.item.variants[b.v].tokens - a.item.variants[a.v].tokens || + (a.item.id < b.item.id ? 1 : -1), + )[0]; + chosen = chosen.filter((c) => c !== worst); + dropped.push(worst.item.id); + } + const overflow = dropped.length > 0; // Greedy fill by value density with per-source diminishing returns. The cut is checked // BEFORE taking an item, and skips only that source — other sources keep competing // (a `break` here used to end the whole fill, and only after taking a 6th item). @@ -226,14 +352,20 @@ export function assemble( for (const item of optional) { const taken = perSource[item.source] ?? 0; if (SOURCE_DISCOUNT ** taken < SOURCE_VALUE_FLOOR) continue; // 4th+ from this source - const variant = item.variants[0]; - if (used() + variant.tokens > budget) continue; + const v = item.variants[0]; + if (used() + v.tokens + (chosen.length ? sepTokens : 0) > budget) continue; chosen.push({ item, v: 0 }); perSource[item.source] = taken + 1; } - const covered = new Set(chosen.flatMap((c) => c.item.covers)); - const missing = required.filter((r) => !r.resolvable || !covered.has(r.key)); + const block = chosen.map((c) => c.item.variants[c.v].text).join(SEP); + const covered = new Set(chosen.flatMap((c) => c.item.variants[c.v].covers)); + const pending = new Set( + chosen.flatMap((c) => c.item.variants[c.v].pending).filter((k) => !covered.has(k)), + ); + const missing = required.filter( + (r) => !r.resolvable || (!covered.has(r.key) && !pending.has(r.key)), + ); const questions = missing .filter((r) => !r.resolvable) .map((r) => @@ -241,32 +373,57 @@ export function assemble( ? `The task names \`${r.name}\` but the repo doesn't define it — which file implements it (or is it new)?` : `The task names \`${r.name}\` but that file doesn't exist — where should this live?`, ); + for (const c of chosen) { + const t = c.item.variants[c.v].truncated; + if (t) truncated.push({ id: c.item.id, ...t }); + } return { - ok: missing.length === 0, + ok: missing.length === 0 && pending.size === 0 && !overflow, budget, - tokens: used(), + tokens: tokensOf(block), + tokenEstimate: TOKEN_ESTIMATE, + overflow, + ...(overflow ? { dropped } : {}), required: required.map((r) => r.key), covered: [...covered].sort(), + pending: [...pending].sort(), missing: missing.map((r) => r.key), questions, + truncated, selection: chosen.map((c) => ({ id: c.item.id, source: c.item.source, gran: c.item.variants[c.v].gran, tokens: c.item.variants[c.v].tokens, + covers: c.item.variants[c.v].covers, })), - block: chosen.map((c) => c.item.variants[c.v].text).join("\n\n"), + block, }; } /** Human rendering for `forge context`. */ export function renderContext(r) { const lines = ["Forge context — budgeted assembly + completeness gate", ""]; + const state = r.ok ? "COMPLETE" : r.overflow ? "OVER BUDGET — INCOMPLETE" : "INCOMPLETE"; lines.push( - ` budget: ${r.tokens}/${r.budget} tokens · required ${r.required.length} · ${r.ok ? "COMPLETE" : "INCOMPLETE"}`, + ` budget: ${r.tokens}/${r.budget} tokens (${r.tokenEstimate ?? TOKEN_ESTIMATE}) · required ${r.required.length} · ${state}`, ); for (const s of r.selection) lines.push(` + ${s.id} [${s.gran}] ${s.tokens}t`); + if (r.overflow) + lines.push( + "", + ` over budget: even pointers could not fit — dropped ${(r.dropped ?? []).join(", ")}`, + ); + if (r.pending?.length) { + lines.push("", " pending reads (pointed at, not delivered — read before acting):"); + for (const p of r.pending) lines.push(` - ${p}`); + } + for (const t of r.truncated ?? []) + if (t.omitted?.length) + lines.push( + ` ~ ${t.id}: ${t.omitted.length} of ${t.total} omitted (${t.omitted.slice(0, 5).join(", ")}${t.omitted.length > 5 ? ", …" : ""})`, + ); if (r.missing.length) { lines.push("", " missing (computed, not a feeling):"); for (const m of r.missing) lines.push(` - ${m}`); diff --git a/src/cortex.js b/src/cortex.js index 6409ced9..46e3845e 100644 --- a/src/cortex.js +++ b/src/cortex.js @@ -4,7 +4,7 @@ // contradiction. Kept fs-thin and deterministic (day + ids passed in) so it's testable // without any hook wiring. -import { recordLessonEvent, supersedeLessonClaim } from "./ledger_bridge.js"; +import { equivalentLesson, recordLessonEvent, supersedeLessonClaim } from "./ledger_bridge.js"; import { ledgerLessons, mergedLessons } from "./ledger_read.js"; import { recordUse, repoLedger } from "./ledger_store.js"; import { @@ -245,15 +245,45 @@ export function applyDistillation(root, lessonId, distilled) { // supersede below rewrite it there (save() is a ledger-only no-op that still succeeds). const lesson = localLessons(root, 0).find((l) => l.id === lessonId); if (!lesson) return false; - const updated = { + const rewritten = { ...lesson, whatWentWrong: distilled.whatWentWrong, correctedBehavior: distilled.correctedBehavior, }; + // A model rewrite is a PROPOSAL (review F07): unless it is equivalent by the narrow rule + // (same text up to case/whitespace/punctuation, no semantic conflict), the lesson's earned + // standing does not carry over to the new wording — it restarts as a candidate that has to + // earn its own confirmations. What it earned before stays inspectable in provenance and on + // the ledger's superseded parent claim. + const hadEvidence = + (lesson.evidenceCount ?? 0) > 0 || + (lesson.contradictionCount ?? 0) > 0 || + lesson.status === "active"; + const updated = + equivalentLesson(lesson, rewritten) || !hadEvidence + ? rewritten + : { + ...rewritten, + status: "candidate", + evidenceCount: 0, + contradictionCount: 0, + quarantineReconfirms: 0, + lastConfirmedDay: lesson.createdDay ?? lesson.lastConfirmedDay, + provenance: { + ...(lesson.provenance ?? {}), + rewrittenFrom: { + whatWentWrong: lesson.whatWentWrong, + correctedBehavior: lesson.correctedBehavior, + status: lesson.status, + evidenceCount: lesson.evidenceCount ?? 0, + contradictionCount: lesson.contradictionCount ?? 0, + }, + }, + }; const ok = save(root, updated).ok; - // A body rewrite changes the content-addressed claim id — supersede in the ledger - // (mint the distilled claim, carry the evidence over, tombstone the template claim) - // or the lesson's history splits across two disjoint claims. + // A body rewrite changes the content-addressed claim id — supersede in the ledger (mint + // the distilled claim, tombstone the template claim as its parent; evidence carries over + // only for an equivalent rewrite) so the lesson's history stays linked, not split. if (ok) supersedeLessonClaim(root, lesson, updated); return ok; } diff --git a/src/cost_report.js b/src/cost_report.js index 6bcb275d..40fe17db 100644 --- a/src/cost_report.js +++ b/src/cost_report.js @@ -351,7 +351,8 @@ export function estimateSpendFromLogs({ root = process.cwd(), fetchImpl, date } else unpriced.push(model); modelBreakdown.push({ model, - cost, + // An unpriced model's cost is UNKNOWN, not zero (review A10). + cost: pricing ? cost : null, priced: Boolean(pricing), priceSource: pricing ? `${pricing.source}${pricing.basis ? `:${pricing.basis}` : ""}` @@ -362,8 +363,20 @@ export function estimateSpendFromLogs({ root = process.cwd(), fetchImpl, date } cacheReadTokens: u.cacheReadTokens, }); } - modelBreakdown.sort((a, b) => b.cost - a.cost); - return { totalCost, sessions, byModel: modelBreakdown, unpriced }; + modelBreakdown.sort((a, b) => (b.cost ?? -1) - (a.cost ?? -1)); + // Label what this number IS (review A10): an estimate — logged tokens × published + // per-token prices, in USD — not an invoice; complete only when every model was priced; + // and it never includes verifier/test runtime or non-model tool costs. + return { + totalCost, + currency: "USD", + basis: "estimate: logged tokens × published per-token prices (not an invoice)", + complete: unpriced.length === 0, + excludes: ["verifier/test runtime", "tool and API costs outside model calls"], + sessions, + byModel: modelBreakdown, + unpriced, + }; } catch { return null; } diff --git a/src/dash.html b/src/dash.html index 356e15d8..98338514 100644 --- a/src/dash.html +++ b/src/dash.html @@ -3,6 +3,7 @@ + forge dash -

Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory, foresight, and guardrails—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
880ms
End-to-end benchmark
Blast-radius scan
1.10ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
+
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory and foresight—with automatic guardrails on Claude Code—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
851ms
End-to-end benchmark
Blast-radius scan
1.68ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
  /plugin install forgekit
 

Recommended · ambient guards on every prompt

diff --git a/mintlify/cli/substrate.mdx b/mintlify/cli/substrate.mdx index f399e3d9..eb8439b0 100644 --- a/mintlify/cli/substrate.mdx +++ b/mintlify/cli/substrate.mdx @@ -110,9 +110,13 @@ Consequence simulation — predicted breaks + the minimal dry-run test suite for ```bash forge imagine "" -forge imagine "" --run # execute the minimal suite sandboxed +forge imagine "" --run # execute the minimal suite in an isolated git checkout ``` +`--run` executes the suite in an isolated git checkout (a detached-HEAD worktree). It isolates +checkout files only — not network, credentials, your home directory or process permissions — and +it tests the committed baseline, not uncommitted changes. + ## `forge lean` Scope-minimality (M5) — measure the diff's footprint vs what the task asked for. diff --git a/mintlify/introduction.mdx b/mintlify/introduction.mdx index 144c5307..8808cc9c 100644 --- a/mintlify/introduction.mdx +++ b/mintlify/introduction.mdx @@ -1,17 +1,21 @@ --- title: "Forge: the cognitive substrate for AI coding agents" -description: "Forge is the cognitive substrate stateless models are missing — memory, foresight, and guardrails — as native config for every AI coding agent." +description: "A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, and explicit verification around coding agents — as native config for every AI coding tool." --- -**One brain for every AI coding agent.** A large language model is stateless: one -context window, wiped every call. It has no memory of what your team learned, no -foresight about what an edit will break, and no enforced guardrails. Forge +**A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, +and explicit verification around coding agents.** A language model keeps no durable state between +independent calls and sees only a bounded context window, so on its own it does not carry what +your team learned, cannot see the parts of the repository an edit affects unless they are in +context, and cannot enforce rules on itself. Forge (`@codewithjuber/forgekit`) is the **cognitive substrate** — the layer that runs _before_ the model edits code, supplying evidence-referenced, content-addressed memory (we -call it "proof-carrying memory"), heuristic impact foresight, and enforced guardrails — and -a **cross-tool config compiler** that delivers that brain as native config into every tool -at once. Claude Code is the deepest-tested integration; the others receive native config and -MCP tools with less real-world exercise. +call it "proof-carrying memory"), heuristic impact foresight, and guardrails (blocking on +Claude Code) — and a **cross-tool config compiler** that delivers that brain as native config +into every tool at once. Claude Code is the deepest-tested integration and the only one with +automatic hooks; the others receive native config (most also an MCP server entry) with less +real-world exercise. The repository's `docs/INTEGRATIONS.md` lists, per tool, what is emitted, +registered, run automatically and enforced. @@ -30,10 +34,11 @@ MCP tools with less real-world exercise. ## The problem -A large language model is stateless — one context window, wiped every call. +A language model keeps no durable state between independent calls, and it sees only what fits in +its context window. -- It has **no memory** of what your team already learned. -- It has **no foresight** about what an edit will break. +- It does not **carry what your team learned** from one session to the next. +- It does not **see what an edit will affect** unless those files are in its context. - It has **no enforced guardrails** — prose rules get forgotten after a compaction. And every tool wants its own config file (`CLAUDE.md`, `AGENTS.md`, `.cursor/rules`, @@ -42,11 +47,12 @@ things, and the compiler that delivers it into every tool from one source. ## The thesis -A model can't learn from your codebase between calls: its weights are frozen and its -working memory is wiped after every response. Memory, foresight, and self-checking -can't be prompted into it — they have to be supplied from _outside_. That outside layer -is the cognitive substrate. Formally, inference is a fixed function `y = f(x)` with no -state between calls; Forge is the state. +A model with frozen weights does adapt within one context — examples, retrieved facts and +feedback change what it does — but it does not guarantee four things on its own: durable state +across independent calls, context beyond its window, weight updates from outcomes, and reliable +self-verification without external evidence. Forge supplies persistence and external checks from +_outside_ the model. It is one tested way of doing that, not the only possible architecture; +the research programme's dated corrections explain the difference. @@ -75,8 +81,9 @@ state between calls; Forge is the state. content-addressed memory: a claim that carries references to its evidence and is trusted only once independent oracles raise its confidence above a floor. The "proof" is that evidence trail, not a formal proof. -- **Foresight before you break things.** Ask "what does changing `verifyToken` break?" - and get the blast radius from the code graph, including coupled files you never named. +- **Heuristic impact before you break things.** Ask "what does changing `verifyToken` + break?" and get the blast radius from a regex-derived code graph, including coupled files you + never named — it can miss files as well as over-warn. - **Guardrails that can't be forgotten.** Deterministic hooks enforce protected paths, cost budgets, and doom-loop detection — they survive a context compaction. - **Work that finishes end to end.** A completion gate blocks "done" once per session @@ -117,8 +124,10 @@ Forge states its own ceiling everywhere. - **Tests and human corrections always win.** - Forge is **beta**. The core (`init`, `sync`, `substrate`, `impact`, `ledger`, guards) - is tested and in daily use; some flags may change before `1.0`. + Forge is **beta**: releases follow semantic versioning (a breaking change is a new major + version), and "beta" describes maturity — heuristic analyses, advisory checks, and less + real-world exercise outside Claude Code. The core (`init`, `sync`, `substrate`, `impact`, + `ledger`, guards) is tested and in daily use. ## Next steps diff --git a/reports/2026-07-05.md b/reports/2026-07-05.md index f4542af8..8e89e492 100644 --- a/reports/2026-07-05.md +++ b/reports/2026-07-05.md @@ -1,5 +1,11 @@ # Claude Config Radar — 2026-07-05 +> ⚠️ **Archived — historical.** An auto-generated daily report from 2026-07-05, kept for the record. +> Not maintained and not the current state; its action items (including the credential-rotation +> reminders, which mention no secret values) are historical and superseded. See +> **[docs/GUIDE.md](../docs/GUIDE.md)** for today's tooling and **[SECURITY.md](../SECURITY.md)** for +> reporting a live security issue. + **TL;DR** - 🔴 **Security-relevant dep bump**: pgvector **0.8.2** fixes a buffer overflow in parallel HNSW index builds (**CVE-2026-3172**) — your stack uses pgvector, so this is the one worth acting on. - 🟡 **shadcn/ui** made **Base UI the default** component library (docs + new projects) as of July 2026 — escalation of yesterday's "picking Base UI ~2:1" note; Radix not deprecated. diff --git a/reports/benchmarks.md b/reports/benchmarks.md index 7bc8cd59..e27ae4c1 100644 --- a/reports/benchmarks.md +++ b/reports/benchmarks.md @@ -4,7 +4,7 @@ > **a number is an assumption until measured.** Every figure in the generated section below > came from an actual run of `npm run bench` on the machine recorded in the environment > block — no projections, no targets, no numbers copied forward from a different machine. -> Re-run `npm run bench` (≈10 s, node stdlib only) and the generated section is rewritten +> Re-run `npm run bench` (≈20 s, node stdlib only) and the generated section is rewritten > in place with your machine's numbers. ## Methodology @@ -134,8 +134,8 @@ appear side by side below, labeled, and are never blended. "platform": "linux", "arch": "x64", "fsType": "ext2/ext3", - "commit": "d2abfa69fb77531199ffc67c5c076b524af69040 + uncommitted changes", - "date": "2026-09-26T20:14:48.438Z" + "commit": "a56606afd5baebdabee95ad95af626a965557d92 + uncommitted changes", + "date": "2026-09-26T20:35:15.125Z" } ``` @@ -143,24 +143,24 @@ appear side by side below, labeled, and are never blended. | suite | benchmark | median | p95 | runs | notes | |-----------|----------------------------------------------|---------|---------|------|--------------------------------------------------------------| -| atlas | full build (this repo) | 794 ms | 970 ms | 5 | 496 files, 13278 symbols, 36899 edges, cap 20000 not reached | -| atlas | incremental rebuild (unchanged) | 385 ms | 550 ms | 5 | per-file hash cache hit | -| atlas | impact("claimText") (warm adjacency) | 1.10 ms | 1.56 ms | 30 | 57 files impacted | -| ledger | mint+put 1000 claims | 266 ms | 332 ms | 5 | 3,762/s | -| ledger | loadClaims at 1000 claims | 15.1 ms | 15.7 ms | 5 | full state from disk | -| ledger | mergeDirs 2×500-claim replicas (250 shared) | 229 ms | 241 ms | 3 | +250 claims, +313 records | -| ledger | val() over 1000 claims | 0.59 ms | 1.93 ms | 20 | 1,687,379/s (mean val 0.51) | -| reuse | fingerprint 2000 specs | 233 ms | 315 ms | 5 | 8,585/s | -| reuse | lookup exact hit, cold @ 100 artifacts | 1.15 ms | 10.6 ms | 10 | tier=exact | -| reuse | lookup exact hit, warm @ 100 artifacts | 0.51 ms | 4.98 ms | 10 | tier=exact | -| reuse | lookup near hit (LSH), cold @ 100 artifacts | 10.9 ms | 17.2 ms | 5 | tier=near, j=0.98 | -| reuse | lookup miss, cold @ 100 artifacts | 10.6 ms | 15.6 ms | 5 | tier=miss | -| reuse | lookup exact hit, cold @ 1000 artifacts | 3.74 ms | 5.02 ms | 10 | tier=exact | -| reuse | lookup exact hit, warm @ 1000 artifacts | 3.32 ms | 4.08 ms | 10 | tier=exact | -| reuse | lookup near hit (LSH), cold @ 1000 artifacts | 122 ms | 138 ms | 5 | tier=near, j=0.95 | -| reuse | lookup miss, cold @ 1000 artifacts | 131 ms | 141 ms | 5 | tier=miss | -| context | assemble() (this repo, 3-symbol task) | 17.4 ms | 19.7 ms | 10 | 4174/6000 tokens, 9 required, incomplete | -| substrate | substrateCheck (allowBuild, llm off) | 880 ms | 962 ms | 3 | 136 impacted files, route simple | +| atlas | full build (this repo) | 719 ms | 753 ms | 5 | 504 files, 13396 symbols, 37369 edges, cap 20000 not reached | +| atlas | incremental rebuild (unchanged) | 346 ms | 365 ms | 5 | per-file hash cache hit | +| atlas | impact("claimText") (warm adjacency) | 1.68 ms | 2.38 ms | 30 | 61 files impacted | +| ledger | mint+put 1000 claims | 248 ms | 326 ms | 5 | 4,038/s | +| ledger | loadClaims at 1000 claims | 13.6 ms | 15.3 ms | 5 | full state from disk | +| ledger | mergeDirs 2×500-claim replicas (250 shared) | 191 ms | 198 ms | 3 | +250 claims, +313 records | +| ledger | val() over 1000 claims | 0.85 ms | 1.39 ms | 20 | 1,170,474/s (mean val 0.51) | +| reuse | fingerprint 2000 specs | 232 ms | 290 ms | 5 | 8,636/s | +| reuse | lookup exact hit, cold @ 100 artifacts | 1.01 ms | 9.43 ms | 10 | tier=exact | +| reuse | lookup exact hit, warm @ 100 artifacts | 0.67 ms | 5.89 ms | 10 | tier=exact | +| reuse | lookup near hit (LSH), cold @ 100 artifacts | 18.5 ms | 33.0 ms | 5 | tier=near, j=0.98 | +| reuse | lookup miss, cold @ 100 artifacts | 18.2 ms | 25.4 ms | 5 | tier=miss | +| reuse | lookup exact hit, cold @ 1000 artifacts | 3.98 ms | 6.11 ms | 10 | tier=exact | +| reuse | lookup exact hit, warm @ 1000 artifacts | 3.38 ms | 4.14 ms | 10 | tier=exact | +| reuse | lookup near hit (LSH), cold @ 1000 artifacts | 128 ms | 133 ms | 5 | tier=near, j=0.95 | +| reuse | lookup miss, cold @ 1000 artifacts | 123 ms | 130 ms | 5 | tier=miss | +| context | assemble() (this repo, 3-symbol task) | 17.6 ms | 27.0 ms | 10 | 4174/6000 tokens, 9 required, incomplete | +| substrate | substrateCheck (allowBuild, llm off) | 851 ms | 995 ms | 3 | 141 impacted files, route simple | ### Impact-oracle quality (hand-labeled cases, this repo) @@ -168,11 +168,11 @@ appear side by side below, labeled, and are never blended. |---------------|-----------|--------|------|-----------|-------| | normalizeSpec | 0.11 | 1.00 | 0.20 | 18 | 2 | | evalImpact | 0.29 | 1.00 | 0.44 | 7 | 2 | -| isStale | 0.19 | 1.00 | 0.32 | 37 | 7 | -| mergeStates | 0.15 | 1.00 | 0.27 | 26 | 4 | -| claimText | 0.18 | 1.00 | 0.30 | 57 | 10 | -| contentHash | 0.11 | 1.00 | 0.20 | 100 | 11 | -| mean of 6 | 0.17 | 1.00 | 0.29 | | | +| isStale | 0.17 | 1.00 | 0.30 | 40 | 7 | +| mergeStates | 0.14 | 1.00 | 0.25 | 28 | 4 | +| claimText | 0.18 | 1.00 | 0.31 | 61 | 11 | +| contentHash | 0.10 | 1.00 | 0.19 | 105 | 11 | +| mean of 6 | 0.17 | 1.00 | 0.28 | | | Edited-file-only baseline recall over the same cases: **0.26**. @@ -183,11 +183,13 @@ derivations, so the rows are comparable in spirit only and are never blended: |------------------------------------------------|-----------|--------|------|------------------------------------------------------------| | paper prototype, self-built demo (REFUTED) | 0.63 | 1.00 | 0.75 | mutation testing on the authors' own fixture | | paper prototype, field study (pooled, 9 repos) | 0.40 | 0.02 | 0.04 | 759 files' mined co-change (research/empirical-refutation) | -| this repo (regex atlas, hand-labeled) | 0.17 | 1.00 | 0.29 | 6 hand-labeled cases (bench/impact_cases.mjs) | +| this repo (regex atlas, hand-labeled) | 0.17 | 1.00 | 0.28 | 6 hand-labeled cases (bench/impact_cases.mjs) | -> **Snapshot boundary.** The measured results above were generated at commit `eb68ea9` and +> **Snapshot boundary.** The measured results above were generated at the commit recorded in +> the environment block (`commit`; "+ uncommitted changes" means the working tree the review +> fixes of 2026-09-26 were measured in, before they were committed) and > do not benchmark the optional embedding adapter now implemented in `src/embed.js` and > exercised with a deterministic fake provider in `test/embed.test.js`. MinHash remains the > zero-dependency default and failure fallback. The structural comparisons below describe the @@ -235,6 +237,6 @@ that forgekit structurally does not. ## Reproduce ```sh -npm run bench # ≈10 s; prints the tables and rewrites the generated section above +npm run bench # ≈20 s; prints the tables and rewrites the generated section above npm test # includes a smoke test of the harness's pure helpers (test/bench.test.js) ``` diff --git a/reports/cost-eval.md b/reports/cost-eval.md index 0c66fd25..7171f28f 100644 --- a/reports/cost-eval.md +++ b/reports/cost-eval.md @@ -6,21 +6,30 @@ > saving. The paper's 62 % routing saving (paper §9) was measured on the 30 tasks its > thresholds were tuned on and is **refuted**: on 80 held-out tasks, counting every escalation, > routing cost 20.2 % _more_ than always-premium ([research/empirical-refutation/](../research/empirical-refutation/)). -> The plan's ~90 % composed figure is a **target**, not a result, and does not appear in this table. +> The plan's ~90 % figure is a **target** and a hypothesis, not a result, and does not appear in this table. +> +> **Corrected 2026-09-26.** The methodology below used to say the cost model "is multiplicative — +> `C = C₀ · Π(1 − fᵢ)` over independent stages — so each stage factor is measured separately and +> composed arithmetically", and that paired runs reprice "identical tokens". Stage savings interact +> (cache hits change the routed workload, context changes retries, halts can defer work), so the +> per-stage factors are diagnostics, not a total; and repricing tokens at another model's price is a +> counterfactual, not an observed outcome. The acceptance rule for any cost headline is in +> [05-cost-model.md §3](../docs/plans/substrate-v2/05-cost-model.md#3-acceptance-rule-for-any-cost-headline). ## Methodology -The cost model is multiplicative — `C = C₀ · Π(1 − fᵢ)` over independent stages — so each -stage factor is measured separately and composed arithmetically, never asserted: +Each stage factor is measured separately as a diagnostic; the system is judged only on paired, +full-system outcomes (total cost per completed, externally verified task), never on a product of +stage factors: 1. **Instrumentation.** Every substrate stage appends one line to `.forge/metrics.jsonl` (`{t, stage, outcome, tokensIn, tokensOut, tier, savedEstimate, ref}` — `src/metrics.js`). `forge cost --stages` computes the per-stage factors from those lines (`src/cost_report.js`); a stage with no events reports **no data**, never a default. -2. **Paired runs.** Baseline (always-premium, read-everything, no cache) vs. substrate over - the same replay corpus (N ≥ 100 real tasks, stratified repeat-heavy / mixed / cold), the - paper §9 methodology: identical tokens repriced, so every saving is arithmetic on measured - tokens. +2. **Paired runs.** Baseline (equivalent tools, context and repair opportunity; always-premium / + read-everything reported too) vs. substrate over the same replay corpus (N ≥ 100 real tasks, + stratified repeat-heavy / mixed / cold), each policy actually executed. Repriced tokens are a + labelled counterfactual, not a measured saving. 3. **Correctness guard (spec §3).** A saving counts only if the external verifier passes the output. A routed-down answer that fails is not a saving; a cache hit that gets reverted is recorded as a *negative* entry. @@ -33,7 +42,7 @@ stage factor is measured separately and composed arithmetically, never asserted: | cache (reuse, tier-weighted) | — | 0 | no data yet — run with metrics enabled | | route (vs always-premium) | — | 0 | no data yet — run with metrics enabled | | context (assembly ρ) | — | 0 | no data yet — run with metrics enabled | -| **composed (measured stages only)** | — | 0 | nothing to compose yet | +| **composed (measured stages only; diagnostic, not a total)** | — | 0 | nothing to compose yet | Secondary counters (doom-loop halts avoided, M5 lean, avoided rework) are reported alongside when populated — they are deliberately excluded from the multiplication (spec §1). diff --git a/research/HISTORICAL_EDITIONS.md b/research/HISTORICAL_EDITIONS.md new file mode 100644 index 00000000..0ad05f00 --- /dev/null +++ b/research/HISTORICAL_EDITIONS.md @@ -0,0 +1,120 @@ +# Historical editions of the research papers + +> ⚠️ **Historical, pre-correction editions.** Every PDF listed here predates the 2026-09-21 and +> 2026-09-26 corrections. It is kept so that what was published can still be read and cited, not +> as the current text. The corrected sources are the HTML and LaTeX files named in the table; +> read those. + +## Status on 2026-09-26: not regenerated + +A re-render of the three HTML papers was attempted on 2026-09-26 and **not committed**: + +1. **Figures need resolving.** The HTML sources reference every figure through a + `{{artifact:…}}` placeholder, which no browser resolves, so a plain render has no figures. The + map below resolves them; it was checked against the images embedded in the old PDFs. +2. **The Qur'anic text could not be verified.** All three HTML papers carry Qur'anic Arabic. When + the white paper and the synthesis were rendered in the environment available that day, Chromium + set their verse text in three fallback fonts at once (DejaVu Sans for most glyphs, Liberation + Serif and FreeSerif for the rest); mixing fonts inside a word can break letter joining and mark + placement, and the rendered pages could not be inspected by eye. A PDF whose sacred text has not + been checked is not published as the new edition. +3. **The refutation paper needs TeX.** `empirical-refutation/paper.pdf` is built from + `paper/main.tex` with a TeX Live toolchain; none was available. + +## The editions + +Each edition stays retrievable byte for byte at the pinned commit +`d2abfa69fb77531199ffc67c5c076b524af69040`: +`git show d2abfa69fb77531199ffc67c5c076b524af69040: > edition.pdf`, or +`git cat-file -p ` with the blob below. + +| PDF (historical, pre-correction) | Git blob at `d2abfa6` | sha256 (prefix) | Bytes | Last changed in | Corrected source | +| --- | --- | --- | --- | --- | --- | +| `research/formal-synthesis/substrate_synthesis.pdf` | `2e17362fc62d9f32b1083f17a0ac865704244ef6` | `644e28d0f8d1cbd3…` | 835072 | `5e60069` (2026-08-14) | `research/formal-synthesis/substrate_synthesis.html` | +| `research/empirical-refutation/extended_preprint.pdf` | `74f74ae08612bb9dc11038f651f923e0730b8bfc` | `8d2ec1091d17f0ba…` | 791707 | `9ebe256` (2026-09-20) | `research/empirical-refutation/extended_preprint.html` | +| `research/empirical-refutation/paper.pdf` | `f94a727cec7f84ac197057f3f22cde0fb09f28b1` | `a5001d8fc59a1b4a…` | 830746 | `c5fb041` (2026-09-20) | `research/empirical-refutation/paper/main.tex` | +| `research/cognitive-substrate/cognitive_substrate_whitepaper.pdf` | `44ce7bbd4a7bc1e6220f162074c9b473c6287e7b` | `599e626ba24958c6…` | 1816584 | `e6e6de7` (2026-09-20) | `research/cognitive-substrate/cognitive_substrate_whitepaper.html` | +| `docs/cognitive-substrate/cognitive_substrate_whitepaper.pdf` (byte-identical copy) | same blob as the row above | same | 1816584 | — | the same HTML, copied to `docs/cognitive-substrate/` | + +The copies of `repro/paper/main.tex` and `repro/paper/paper.pdf` inside +`empirical-refutation/replication_package.tar.gz` (git blob `50bd453a30dad5d8ca3369129f8015fd4524fa81`) +are also left exactly as published. The paper PDF was built with pdfTeX (TeX Live 2026) and the ACM +`acmart` class, as its own metadata records. + +## Figure map + +Each `{{artifact:}}` placeholder in the HTML sources (13 in all: 3 in the synthesis, 3 in the +preprint, 7 in the white paper), the figure file under `research/` it stands for, and whether that +file's pixel size matches the image embedded in the old PDF: + +| Placeholder id | Figure (under `research/`) | Size (px) | Matches the old PDF | +| --- | --- | --- | --- | +| `art_5f049677-4c3c-40f2-8905-dd01c966e9ae` | `formal-synthesis/figures/schematic_duality.png` (synthesis and preprint, Figure 1) | 1366 × 1046 | yes | +| `art_f0decf80-d016-496f-8032-f7b2e73e71b2` | `formal-synthesis/figures/schematic_taskloop.png` (synthesis and preprint, Figure 2) | 1607 × 1092 | yes | +| `art_5d076ce3-0f54-4394-9a77-f70a336ca843` | `formal-synthesis/figures/schematic_convergence.png` (synthesis, Figure 8) | 2460 × 1539 | yes | +| `art_712fac51-fe17-4ed6-80f0-9dd42bf42758` | `empirical-refutation/figures/fig_repair_beforeafter.png` (preprint, Figure 3) | 3142 × 1383 | yes | +| `art_e2776474-3d1e-48c6-9490-55d4d927a301` | `cognitive-substrate/figures/schematic_loop.png` (white paper, Figure 1) | 3003 × 1439 | yes | +| `art_d2be1b53-86ce-4069-b2aa-5be59836598e` | `cognitive-substrate/figures/schematic_system.png` (white paper, Figure 2) | 2847 × 1840 | yes | +| `art_8d9fa6dd-3554-49c7-9e76-ba667544a622` | `cognitive-substrate/figures/schematic_extended.png` (white paper, Figure 3) | 1483 × 931 | yes | +| `art_07bb9186-5e55-44f6-af3f-dde83d6b9e65` | `cognitive-substrate/figures/impact_graph.png` (white paper, Figure 4) | 1900 × 1326 | yes | +| `art_392e293d-be93-4efe-81c1-e9612a711ac4` | `cognitive-substrate/figures/eval_precision_recall.png` (white paper, Figure 5) | 2300 × 918 | **no** — the old PDF embeds a 1921 × 842 raster, so the repository's file is a different render of this figure; compare the two before publishing | +| `art_5b206b7b-c90e-417f-b1e8-48b0ec389cb8` | `cognitive-substrate/figures/schematic_router_loop.png` (white paper, Figure 6) | 1537 × 838 | yes | +| `art_ac78be07-be03-4560-bb9f-f5fe2f16d7ef` | `cognitive-substrate/figures/router_eval.png` (white paper, Figure 7) | 1719 × 732 | yes | + +## Rendering a new edition + +1. Work in a scratch directory outside the repository and install `playwright-core` there, never + in the repository: `npm init -y && npm install playwright-core`. Point it at an installed + Chromium (`executablePath`). +2. Install a font with full Qur'anic coverage (a Naskh face such as Amiri or Scheherazade New) and + make it the first `font-family` for `.quran .ar` in the render, so one font sets each verse. +3. Replace every `{{artifact:}}` with its figure from the map (a `data:image/png;base64,…` + URI keeps the render self-contained), render A4 with a header and footer stamp + `edition · source sha256 · forgekit `, + and confirm every image loaded and no request failed. +4. **Inspect by eye** every page that carries a figure or a verse card. Only then replace the PDF, + re-copy the white paper to `docs/cognitive-substrate/` (`node scripts/claims-status.mjs + --sync-copies`), and add a row here recording the replaced edition's git blob and the new + edition's source hash. + +A render script that does steps 1 and 3 (it resolved all 13 figure placeholders across the three +papers with no failed request on 2026-09-26): + +```js +// node render.mjs (run from the scratch directory) +import { createHash } from "node:crypto"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { chromium } from "playwright-core"; + +const FIGURES = { /* "art_…": "research/…/figures/….png", one entry per row of the map above */ }; +const [repo, rel, out] = process.argv.slice(2); +const raw = readFileSync(path.join(repo, rel)); +const sha = createHash("sha256").update(raw).digest("hex").slice(0, 12); +const version = JSON.parse(readFileSync(path.join(repo, "package.json"), "utf8")).version; +const stamp = `edition ${new Date().toISOString().slice(0, 10)} · source sha256 ${sha} · forgekit ${version}`; +let html = raw.toString("utf8"); +for (const [id, fig] of Object.entries(FIGURES)) { + const uri = `data:image/png;base64,${readFileSync(path.join(repo, fig)).toString("base64")}`; + html = html.split(`{{artifact:${id}}}`).join(uri); +} +const browser = await chromium.launch({ executablePath: process.env.CHROMIUM }); +const page = await browser.newPage(); +const failed = []; +page.on("requestfailed", (r) => failed.push(r.url())); +await page.setContent(html, { waitUntil: "networkidle" }); +const unresolved = (html.match(/\{\{artifact:[^}]+\}\}/g) || []).length; +const broken = await page.evaluate(() => [...document.images].filter((i) => !i.naturalWidth).length); +if (unresolved || broken || failed.length) throw new Error(`figures: ${unresolved} unresolved, ${broken} broken, ${failed.length} failed`); +const line = (s) => `
${s}
`; +await page.pdf({ + path: out, format: "A4", printBackground: true, displayHeaderFooter: true, + margin: { top: "18mm", bottom: "18mm", left: "14mm", right: "14mm" }, + headerTemplate: line(stamp), + footerTemplate: line(`${stamp} · page / `), +}); +await browser.close(); +``` + +For `paper.pdf`, build `paper/main.tex` with TeX Live (`pdflatex` and `bibtex`, ACM `acmart` +class) and stamp the same fields in the PDF metadata or a footnote. diff --git a/research/README.md b/research/README.md index 6cac916d..8eea9665 100644 --- a/research/README.md +++ b/research/README.md @@ -1,56 +1,122 @@ # Research -The full research programme behind forgekit: a theory of what a frozen language model -structurally lacks, an architecture that supplies it, two runnable prototypes, and — most -importantly — a pre-registered empirical evaluation that **refuted the prototypes' headline -claims**. +The full research programme behind forgekit: an account of what a language model with frozen +weights does not guarantee on its own, an architecture that supplies it, two runnable +prototypes, and — most importantly — a pre-registered empirical evaluation that **refuted the +prototypes' headline claims**. Read in this order. The later work corrects the earlier work, and the corrections are the -most useful part. +most useful part. Every load-bearing headline below also has a row, with its status and the +evidence behind it, in the machine-readable claim registry +[`docs/status/claims.json`](../docs/status/claims.json), rendered as a table in +[`docs/status/README.md`](../docs/status/README.md). ## Start here: what is actually true | | Claimed (self-built demos) | Measured (real data) | |---|---|---| | Impact oracle recall | 1.00 | **0.022** — `grep` with no graph beats it ~10× on F1 | -| Router/gate F1 | 1.00 | **0.37** on 80 real GitHub issues/PRs | -| Cost saving | +62.1% | **−20.2%** — routing costs *more* than always-premium | +| Router/gate: gate F1 (should-ask) | 1.00 | **0.37** on 80 real GitHub issues/PRs | +| Router/gate: cost saving vs always-premium | +62.1% | **−20.2%** — routing costs *more* than always-premium | -Per output a judge accepted, the router cost $1.06 against always-premium's $1.76, but -only 6 and 3 of 64 outputs were accepted, so that comparison is not stable; 58 of the 64 -tasks failed at every tier, which is why escalation made routing cost more overall. +Success in the router rows means **judge-accepted**: a model judge accepted the output. No +held-out task admitted execution-based verification, so `tests_passed`, `human_accepted` and +`deployed_without_revert` were never measured. Per judge-accepted output the router cost $1.06 +against always-premium's $1.76, but only 6 and 3 of the 64 non-halted tasks were judge-accepted, +so that ratio is not stable; 58 of the 64 tasks failed at every tier, which is why escalation made +routing cost more overall ($6.3582 against $5.2893, 20.21% more). The judge was also the mid-tier +executor, and the "second labelling pass" is the same model with a reworded prompt (n = 30: halt +κ 0.5161, tier κ 0.8919), so κ measures self-consistency, not agreement with a human. +(Corrected 2026-09-26: the first sentence read "Per output a judge accepted, …" and named neither +what the judge's acceptance is not nor who the judge was; the row was labelled "Router/gate F1".) -After diagnosing and repairing two defects, with numeric parameters frozen before the -held-out repositories were touched: recall **0.653**, F1 **0.416**, a point estimate above -`grep`'s 0.371 for the first time. That the repaired oracle *beats* grep is **not -established**: the three held-out repositories all favour it, but three out of three is a -one-sided sign-test p of 0.125, pytest supplies 71% of the held-out pairs, the file-level -intervals overlap, and the choice of which relations to add was made on all nine -repositories. (Corrected 2026-09-21; earlier versions called it "a real but narrow win".) +After diagnosing and repairing two defects, with numeric parameters frozen before the held-out +repositories were touched: at the pre-registered canonical threshold 0.02, recall **0.653** and +F1 **0.416**, a point estimate above `grep`'s 0.371 for the first time (paired ΔF1 about +0.044). +That the repaired oracle *beats* grep is **not established**. Per-repository counts exist only at +threshold 0.10, where the pooled ΔF1 is +0.0565 and all three held-out repositories favour the +oracle, but three out of three is a one-sided sign-test p of 0.125; pytest supplies 71.3% of the +held-out pairs; the file-level intervals overlap; and the choice of which relations to add was made +after diagnosing all nine repositories — an architecture-selection channel into the nominal test +set, which limits the unseen-repository claim without erasing the measured gain. Numbers at 0.02 +and 0.10 are never mixed in one comparison. (Corrected 2026-09-21; earlier versions called it "a +real but narrow win". Thresholds separated 2026-09-26.) The general lesson, demonstrated on our own work: **a self-built demonstration can overstate field performance by more than an order of magnitude, and careful caveating does not convert a demonstration into evidence.** +## What the impact study measured — and what it did not + +*(Added 2026-09-26, after a second external review recomputed the archived results.)* + +The archived counts reproduce: 801 labelled files, 759 evaluated after the pre-registered cap of +200 files per repository, nine repositories, 20,144 mirrored labelled pairs. The original oracle's +pooled precision / recall / F1 is **0.3982 / 0.0220 / 0.0416**; grep's is **0.3535 / 0.5732 / +0.4373**. A repository-cluster bootstrap (20,000 draws, seed 1234) gives oracle F1 **[0.0010, +0.0927]**, grep **[0.3807, 0.5394]**, and grep minus oracle **[0.3422, 0.5174]** — figures +recomputed by the 2026-09-26 external review and reproduced by +[`recompute_corrections.py`](recompute_corrections.py) §5, which also prints the repository-level +view: macro F1 0.0220 for the oracle against 0.4947 for grep, with grep ahead in 9 of 9 +repositories. The negative result is well supported within this archived corpus. + +What it is a result *about* needs stating as carefully as the numbers: + +- **Co-change is a proxy.** Two files that changed in the same commit are *historically related + edits*, not proof of semantic necessity or of test breakage; conversely a dependency graph is not + a full co-change graph. The study measured one task: (a) predicting co-edited files. The other + task an impact tool is used for — (b) selecting the tests that detect a behaviour regression — + was not measured, and results for the two should always be reported separately. +- **Pairs are not independent.** Every ground-truth pair is mirrored (counted from both ends) and + files share repositories, so file- or pair-level resampling overstates precision. Uncertainty is + reported at the repository level, and macro results sit beside pooled ones. +- **The Node graph is not the evaluated oracle.** The shipped `forge impact` / `src/atlas.js` is a + regex-derived, multi-language code graph that ports the two repairs; it is not the Python AST + oracle the study evaluated, and the study's numbers are not its numbers. Its own measurement is a + six-case, self-labelled fixture in [`reports/benchmarks.md`](../reports/benchmarks.md). + +### Next study (pre-declared shape) + +The nine-repository archive is a reproducibility starter, not a fresh holdout. The next impact +study freezes the parser and relation design **before** acquiring a new repository set or time +split; includes runtime coupling, configuration changes, dynamic imports and languages other than +Python; predeclares relation budgets so that widening predictions cannot win merely by returning +most files; and reports review-cost metrics — files reviewed per true affected file, and the +missed-regression rate — beside F1, for co-edited-file prediction and regression-test selection +separately. + ## The four layers ### 1. [`cognitive-substrate/`](cognitive-substrate/) — the theory -The originating argument: an LLM is a frozen map `y = f_θ(x)` with three properties — -statelessness, frozen parameters, bounded context — which structurally deny it five faculties -(memory, learning, imagination, self-correction, impact-awareness). The remedy is an external -stateful architecture, not better prompting. +The originating argument: an LLM is a map `y = f_θ(x)` with frozen parameters, no state between +calls and a bounded context, and a coding agent built on it lacks five faculties (memory, +learning, imagination, self-correction, impact-awareness) unless something outside supplies them. +Stated precisely, what is missing is a set of guarantees — no durable state across independent +invocations, a bounded context, no automatic parameter update, and unreliable self-verification +without external evidence. Prompting does change behaviour inside a context (in-context adaptation; +Brown et al., 2020, [arXiv:2005.14165](https://arxiv.org/abs/2005.14165)); the substrate is a tested +way of supplying persistence and verification, not the only logically possible architecture. +(Corrected 2026-09-26: this paragraph said the three properties "structurally deny it five +faculties" and that "the remedy is an external stateful architecture, not better prompting".) -- `cognitive_substrate_whitepaper.pdf` — the *Theory → Evidence → Build-Map* edition (48pp); - the `.html` edition carries the 2026-09-21 corrections, the PDF predates them +- `cognitive_substrate_whitepaper.pdf` — the *Theory → Evidence → Build-Map* edition (48pp). + **Historical, pre-correction edition** (git blob `44ce7bb`); the `.html` edition is the corrected + source and carries the 2026-09-21 and 2026-09-26 corrections. - `EXECUTIVE_SUMMARY.md` — one-page entry point, **carries a status banner: its prototype numbers are refuted** - `literature/` — the gap map and 32 graded references behind each faculty claim - `evidence/` — twelve load-bearing industry statistics independently re-grounded and graded `confirmed` / `vendor-reported` / `unverifiable`, plus an ecosystem map of what the 2026 Claude-Code stack already solves. Three widely-repeated statistics were caught as - misattributed and dropped. + misattributed and dropped. Since 2026-09-26 the evidence map also grades claim support, study + design, independent replication and transfer scope separately; the original grades mainly + confirm that a source exists and says what is quoted. - `quranic-lens/` — the fourteen-mapping ethical-epistemic reading used as a *design lens*: it names which safeguards are obligatory rather than optional. It is framing, never - technical authority; no verse is offered as proof of an engineering claim. + technical authority; no verse is offered as proof of an engineering claim. The Arabic source + text, the translation, tafsir and the author's design analogy are labelled separately, and the + lens's operational content is the discipline *do not assert without evidence* — the claim + registry, verifier events and visible uncertainty — not any algorithm's correctness, catch rate + or uniqueness. - `sources/` — the primary documents the evidence layer was graded against - `figures/` — the architecture schematics and prototype evaluations @@ -62,7 +128,11 @@ two-layer duality: the silent-miss residual is `(1 − p) × P(no deterministic check fires | miss)`, so where each factor is bounded away from zero, neither layer alone reaches a small residual. Since the 2026-09-21 corrections this is stated as a bound over an explicit `(p, q)` region, not as a proof that -neither layer suffices, and the checks multiply only if they fire independently. +neither layer suffices, and the checks multiply only if they fire independently. Since the +2026-09-26 corrections the reachable residual is a minimum over the *jointly* feasible `(p, q)` +pairs — separately maximal `p` and `q` need not be attainable under one policy, so +`(1 − p_max)(1 − q_max)` is only a lower bound — and a caught miss is no longer read as a +completed task. **Priority note:** prior-art review found this composition law is standard protection-layer algebra, and two concurrent preprints derive a strictly more general Bayesian form weeks @@ -70,6 +140,9 @@ earlier. Priority is conceded in the refutation paper's related work and, since 2026-09-21 corrections, in the synthesis and the extended preprint as well (before that they still said "this paper proves"). What survives is that both preprints are simulation-only. +- `substrate_synthesis.pdf` — **historical, pre-correction edition** (git blob `2e17362`); the + corrected source is `substrate_synthesis.html`. + ### 3. [`empirical-refutation/`](empirical-refutation/) — the measurement The pre-registered evaluation that overturned the claims above, the diagnosis of *why*, and the repair. Includes a replication package with the frozen pre-registration, mined ground @@ -81,10 +154,44 @@ Also corrects a theoretical claim: perfect recall was inferred from a completene but such a theorem guarantees completeness only *relative to the relation* the closure runs over — it says nothing about whether that relation contains the edges that matter. +- `replication_package.tar.gz` — the archive **exactly as published** (git blob `50bd453`); its + copies of `paper/main.tex` and `paper.pdf` predate the corrections. Corrected summary: the + README's Corrections sections; every corrected number is recomputed from it by + `recompute_corrections.py`. +- `paper.pdf` and `extended_preprint.pdf` — **historical, pre-correction editions** (git blobs + `f94a727`, `74f74ae`); the corrected sources are `paper/main.tex` and `extended_preprint.html`. + ### 4. [`python-prototypes/`](python-prototypes/) — the code -`impact_oracle/` and `router_gate/`, runnable with their own test suites. The **repaired** -oracle ships inside the refutation's replication package rather than replacing the version -here, so swapping it in stays a deliberate decision. +`impact_oracle/` and `router_gate/`, runnable with their own test suites. The in-tree +`impact_oracle/` **is the repaired (v2) oracle**: both repairs are in its source, with their +frozen parameters as module defaults, and its suite is 49 tests (36 demo-package tests plus 13 +regression tests for the two repairs). `ImpactOracle(wm, sibling_enabled=False, +forward_enabled=False)` reproduces the refuted reverse-only traversal, and the untouched as-shipped +v1 package is archived in the replication tarball. (Corrected 2026-09-26: this paragraph said the +repaired oracle "ships inside the refutation's replication package rather than replacing the +version here, so swapping it in stays a deliberate decision"; the swap had already been made.) + +## Prior art, and what is (and is not) claimed + +*(Added 2026-09-26.)* External memory, feedback-driven improvement and structured agent control +all have clear prior art. **CoALA** (Sumers et al., 2023, +[arXiv:2309.02427](https://arxiv.org/abs/2309.02427)) organises language agents into modular +memory, action and decision procedures; **Reflexion** (Shinn et al., 2023, +[arXiv:2303.11366](https://arxiv.org/abs/2303.11366)) improves agents through linguistic feedback +and an episodic memory buffer, with no weight updates; GPT-3's few-shot evaluation +([arXiv:2005.14165](https://arxiv.org/abs/2005.14165)) already measured adaptation through text +alone. That prior art does not make forgekit unoriginal as a product, but it limits what the broad +architecture can claim. The defensible framing is: + +> **a portable implementation of evidence-weighted coding-agent memory and checks, with +> empirical evaluation of trust failure modes.** + +Novelty is claimed only for a specific protocol, invariant, evaluation result or integration that +survives an explicit comparison with that prior art. The "five faculties" are a useful +decomposition, not a proof that these five are necessary or that an external stateful architecture +is the only way to supply them; and the "convergence" of the theory, forgekit and its sibling +projects is consistency within one author's work, not independent confirmation — the same care the +programme already applied when it conceded priority for the protection-layer equation. ## How this programme tries to stay honest @@ -98,6 +205,15 @@ Where this falls short is stated too: the pre-registration and parameter freezes self-administered with no external timestamping authority, so a reader can verify internal consistency and the amendment trail but must take the ordering on trust. +### Four kinds of reproducibility, kept apart + +| Kind | Status | +|---|---| +| **Source availability** | The papers' corrected sources (HTML, LaTeX), both Python prototypes, the replication archive and the recomputation script are all in this directory. | +| **Calculation reproducibility** | Available. [`recompute_corrections.py`](recompute_corrections.py) (standard library only) recomputes every corrected statistic from the archived results, and asserts the Theorem D sanity checks with no data at all (`--theorem-checks`); CI runs both, and both prototypes' test suites, since `aedddf5`. The universal router's shipped prior also refits exactly from pinned public data with `bench/universal-router/reproduce.sh` (2026-09-26, the project's own run). | +| **Pipeline reproducibility** | **Not available.** The historical mining pipeline (cloning, commit filtering, labelling, model calls) is not shipped as an entry point here, so new histories cannot be mined with one command; and the universal router's run-4 held-out benchmark ran in an external harness (harness-bench) that is not shipped either — see [`docs/UNIVERSAL_ROUTING.md`](../docs/UNIVERSAL_ROUTING.md). | +| **Independent external replication** | None of the research results has been replicated by an independent team. The 2026-09-21 and 2026-09-26 reviews recomputed archived numbers; they did not re-mine repositories or re-run model calls. | + ## Corrections (2026-09-21) An external deep review of this repository (2026-09-21) recomputed the research statistics @@ -126,10 +242,40 @@ mkdir rp && tar -xzf research/empirical-refutation/replication_package.tar.gz -C python research/recompute_corrections.py rp/repro ``` -**Stale PDFs.** These PDFs predate the corrections and could not be rebuilt here (the HTML -editions were rendered with WeasyPrint, the paper with a TeX Live toolchain; neither was -available): `formal-synthesis/substrate_synthesis.pdf`, +## Corrections (2026-09-26) + +A second external deep review (2026-09-26, pinned at commit +`d2abfa69fb77531199ffc67c5c076b524af69040`) recomputed the archived results again and read the +papers' framing against the code. The counts reproduced exactly again. Changes, each marked in +place in its paper with `[corrected 2026-09-26]` and listed there with the original wording: + +- **Formal synthesis and extended preprint** — the range statement of Theorem D no longer combines + separately maximal `p` and `q` (counterexample: policies `(0.5, 0.9)` and `(0.9, 0.1)` leave 0.05 + and 0.09, while the separate maxima suggest 0.01); the equality condition for + `1 − (1 − ε)ⁿ` is every `rᵢ = ε`, not independence alone; a new §5.4 separates silent-miss + probability from completed-task rate and lists what to measure; the frozen-map premise is stated + as the guarantees it removes; prior art (CoALA, Reflexion) is named and the byline no longer + calls the three bodies of work "independently-developed". The counterexample, the equality + condition and the 400× correction are asserted by `python3 research/recompute_corrections.py + --theorem-checks`. +- **Whitepaper** — the "cannot learn / imagine / self-correct" framing marked as broader than the + missing guarantees; prior art added to §11; the Qur'anic lens's text, translation, tafsir and + design analogy labelled separately, with its operational scope stated; METR's 19% slowdown + scoped to its 16 developers, 246 tasks and early-2025 tools, with METR's + [February 2026 update](https://metr.org/blog/2026-02-24-uplift-update/). +- **This README and the prototype READMEs** — the repaired oracle's location, the 0.02 / 0.10 + thresholds, judge-accepted versus executed success, the unit of the impact study, and the + reproducibility table above. + +**Stale PDFs — historical, pre-correction editions.** `formal-synthesis/substrate_synthesis.pdf`, `empirical-refutation/extended_preprint.pdf`, `empirical-refutation/paper.pdf`, -`cognitive-substrate/cognitive_substrate_whitepaper.pdf`, and the copy in -`docs/cognitive-substrate/`. The copies of `paper/main.tex` and `paper.pdf` inside -`replication_package.tar.gz` are left as published. Read the HTML and LaTeX sources. +`cognitive-substrate/cognitive_substrate_whitepaper.pdf` and its byte-identical copy in +`docs/cognitive-substrate/` predate both sets of corrections. Read the HTML and LaTeX sources, +which carry them. A re-render was attempted on 2026-09-26 and **not** committed: the HTML sources +reference their figures through `{{artifact:…}}` placeholders that a browser cannot resolve, all +three HTML papers carry Qur'anic Arabic whose typesetting could not be checked by eye in that +environment, and a PDF whose figures or sacred text cannot be verified is not published as the new +edition. The paper PDF needs a TeX toolchain that was not available. Each edition's git blob, the +pinned commit where it stays retrievable, and a faithful render recipe (including the +figure-placeholder map) are in [`HISTORICAL_EDITIONS.md`](HISTORICAL_EDITIONS.md). The copies of +`paper/main.tex` and `paper.pdf` inside `replication_package.tar.gz` are left as published. diff --git a/research/cognitive-substrate/EXECUTIVE_SUMMARY.md b/research/cognitive-substrate/EXECUTIVE_SUMMARY.md index 0938a0af..f6d12774 100644 --- a/research/cognitive-substrate/EXECUTIVE_SUMMARY.md +++ b/research/cognitive-substrate/EXECUTIVE_SUMMARY.md @@ -7,7 +7,7 @@ > | Claim below | Measured on real data | > |---|---| > | Impact oracle recall **1.00** | **0.022** (9 OSS repos; 801 labelled files, 759 evaluated); `grep` beats it ~10× on F1 | -> | Router/gate F1 **1.00**, cost saving **+62.1%** | F1 **0.37**; cost saving **−20.2%** (routing costs *more* than always-premium; per judged-correct output $1.06 vs $1.76, from only 6 and 3 correct outputs of 64) | +> | Router/gate F1 **1.00**, cost saving **+62.1%** | F1 **0.37**; cost saving **−20.2%** (routing costs *more* than always-premium; per judge-accepted output $1.06 vs $1.76, from only 6 and 3 judge-accepted outputs of 64; the judge is a model, not executed tests) | > > The theory sections remain the programme's working framework. The *numbers* here do not. A repair > raised recall to 0.653 and F1 to 0.416, a point estimate above grep's 0.371, documented in the @@ -25,6 +25,17 @@ **One-line thesis:** The faculties a coding agent lacks — memory, learning, imagination, self-correction, impact-awareness — are not gaps in the model's *knowledge* but structural consequences of what a frozen transformer *is* (a stateless map `y = f_θ(x)`, fixed weights, bounded window). They cannot be prompted or tooled away; they can only be supplied by **re-wrapping the input→process→output loop** into a closed, stateful cycle around the frozen model. +> *Corrected 2026-09-26:* the thesis above is broader than its argument. Frozen weights rule out +> weight updates during use, not all adaptation: examples, retrieved facts and feedback in the +> context change behaviour with no gradient step (Brown et al., 2020, +> [arXiv:2005.14165](https://arxiv.org/abs/2005.14165)). What a bare model lacks is a set of +> guarantees — no durable state across independent invocations, a bounded context, no automatic +> parameter update, and unreliable self-verification without external evidence — and the substrate +> is one tested way of supplying persistence and verification, not the only possible architecture. +> Prior art for the broad architecture includes CoALA ([arXiv:2309.02427](https://arxiv.org/abs/2309.02427)) +> and Reflexion ([arXiv:2303.11366](https://arxiv.org/abs/2303.11366)); see the white paper's +> Corrections (2026-09-26). + **What v2 adds.** The first edition argued the five faculties from first principles and prototyped the one that is buildable today. This edition (1) **grounds the argument in the field's own evidence** — twelve load-bearing pain-point statistics independently re-grounded from primary sources and graded *confirmed / vendor-reported / unverifiable*; (2) adds **six metacognitive mechanisms** the frozen loop also lacks (routing, assumption gate, decomposition, goal-anchoring, anti-over-engineering, inline verification); (3) **maps all eleven capabilities against the real 2026 Claude-Code stack**, marking each solved / partial / residual-gap so we say clearly *what not to build*; and (4) ships a **second runnable prototype** — a complexity-aware router + assumption gate, evaluated live on real models. > **Governing discipline (the user's, adopted throughout):** *AI output is mathematically-calculated probability — non-deterministic, and never blindly trusted.* Every claim in this package is graded by how well it is sourced; every prototype decision is a transparent, attributable rule rather than another opaque model call; and trust is always earned by an **external** check, never asserted by the model. diff --git a/research/cognitive-substrate/cognitive_substrate_whitepaper.html b/research/cognitive-substrate/cognitive_substrate_whitepaper.html index 95460e9b..615495dc 100644 --- a/research/cognitive-substrate/cognitive_substrate_whitepaper.html +++ b/research/cognitive-substrate/cognitive_substrate_whitepaper.html @@ -65,6 +65,7 @@ .quran .map{font-size:.92rem; color:var(--ink); margin-top:.6em; padding-top:.6em; border-top:1px dotted #d8ccae;} .quran .map b{color:var(--quran);} .quran .grounding{font-size:.74rem; color:var(--faint); margin-top:.5em; font-family:monospace;} + .quran .lbl{font-family:sans-serif; font-size:.64rem; text-transform:uppercase; letter-spacing:.08em; color:var(--faint); margin:.55em 0 .1em;} .lit{background:var(--litbg); border:1px solid #d9e2ea; border-radius:4px; padding:6px 14px; margin:1.1em 0; font-size:.9rem;} .callout{border:1px solid var(--rule); border-radius:5px; padding:14px 20px; margin:1.4em 0; background:#fcfcfc;} .callout.key{border-left:3px solid var(--oracle); background:#f2f9f8;} @@ -96,13 +97,13 @@

A Cognitive Substrate for Coding Agents

-
Status: both prototype results in this edition were refuted · corrections 2026-09-21
-

This edition was written before any real-repository evaluation existed. A later pre-registered evaluation (research/empirical-refutation/) overturned both prototype claims: the impact oracle’s recall was 0.022, not 1.00, on 759 files in nine open-source repositories, and on 80 held-out tasks the router’s total spend was 20.2% higher than always using the premium tier, not 62.1% lower. An external review (2026-09-21) also found a misquoted statistic, a wrong worst-case cost, and an inconsistency between Eq. (1) and M2. Corrections are made in place, marked [corrected 2026-09-21] or [refuted], and listed with the original wording in Corrections. The theory sections remain the programme’s working framework; the prototype numbers do not. The PDF edition predates these corrections.

+
Status: both prototype results in this edition were refuted · corrections 2026-09-21 and 2026-09-26
+

This edition was written before any real-repository evaluation existed. A later pre-registered evaluation (research/empirical-refutation/) overturned both prototype claims: the impact oracle’s recall was 0.022, not 1.00, on 759 files in nine open-source repositories, and on 80 held-out tasks the router’s total spend was 20.2% higher than always using the premium tier, not 62.1% lower. An external review (2026-09-21) also found a misquoted statistic, a wrong worst-case cost, and an inconsistency between Eq. (1) and M2. Corrections are made in place, marked [corrected 2026-09-21] or [refuted], and listed with the original wording in Corrections. The theory sections remain the programme’s working framework; the prototype numbers do not. A second review (2026-09-26) found that the “cannot learn / imagine / self‑correct” framing is broader than the missing guarantees it rests on, that the prior art for the architecture needed stating, that the Qur’anic lens mixed source text, translation and the author’s analogy without labels, and that the METR statistic needed its scope; those are listed in Corrections (2026-09-26) and marked [corrected 2026-09-26]. The PDF edition predates both sets of corrections.

Abstract

-

A large language model at inference time is, mathematically, a fixed function y = fθ(x) with frozen parameters θ and a bounded input window. From this single fact, five apparent “cognitive” deficits of a coding agent follow as structural consequences, not incidental weaknesses: it cannot remember across sessions, cannot learn from outcomes, cannot imagine the consequences of an action before taking it, cannot reliably correct itself, and does not know what already exists in a codebase or what an edit will affect. We show that neither better prompting nor additional tools (skills, MCP servers) remove these deficits, because they leave fθ and the open‑loop pipeline intact. We then specify a cognitive substrate: an external architecture that keeps the LLM frozen but re‑wraps its input→process→output loop into a closed, stateful cycle over persistent stores — an episodic/semantic memory, an online‑updatable learning layer, a consequence simulator, a metacognitive verification gate, and a persistent structural model of the codebase — all under an explicit stewardship boundary. For each faculty we identify precisely what the existing literature solves and what residual gap remains for a coding agent. To turn the weakest‑evidenced claim into something testable, we build and evaluate the impact‑awareness faculty as a runnable prototype: a Codebase World‑Model that parses a repository into a persistent dependency graph, and an Impact Oracle that predicts the blast radius of a proposed edit. Against mutation‑derived ground truth on a ten‑file package we wrote, the oracle was the only method that missed no affected file (recall = 1.00 across five tested edits), where a text‑search baseline missed transitive dependents and an edited‑file‑only baseline missed 47% of impact. On nine real repositories its recall was 0.022, and text search beat it by an order of magnitude on F1. [refuted — see Corrections] Throughout, a Qur'anic epistemic lens supplies the design's vocabulary of obligation — know what exists before acting (2:31–32), verify before you act (49:6), pursue not that of which you have no knowledge (17:36), and hold what you can damage as a trust (33:72).

+

A large language model at inference time is, mathematically, a fixed function y = fθ(x) with frozen parameters θ and a bounded input window. From this single fact, five apparent “cognitive” deficits of a coding agent follow as structural consequences, not incidental weaknesses: it cannot remember across sessions, cannot learn from outcomes, cannot imagine the consequences of an action before taking it, cannot reliably correct itself, and does not know what already exists in a codebase or what an edit will affect. We show that neither better prompting nor additional tools (skills, MCP servers) remove these deficits, because they leave fθ and the open‑loop pipeline intact. [corrected 2026-09-26 — see Corrections] We then specify a cognitive substrate: an external architecture that keeps the LLM frozen but re‑wraps its input→process→output loop into a closed, stateful cycle over persistent stores — an episodic/semantic memory, an online‑updatable learning layer, a consequence simulator, a metacognitive verification gate, and a persistent structural model of the codebase — all under an explicit stewardship boundary. For each faculty we identify precisely what the existing literature solves and what residual gap remains for a coding agent. To turn the weakest‑evidenced claim into something testable, we build and evaluate the impact‑awareness faculty as a runnable prototype: a Codebase World‑Model that parses a repository into a persistent dependency graph, and an Impact Oracle that predicts the blast radius of a proposed edit. Against mutation‑derived ground truth on a ten‑file package we wrote, the oracle was the only method that missed no affected file (recall = 1.00 across five tested edits), where a text‑search baseline missed transitive dependents and an edited‑file‑only baseline missed 47% of impact. On nine real repositories its recall was 0.022, and text search beat it by an order of magnitude on F1. [refuted — see Corrections] Throughout, a Qur'anic epistemic lens supplies the design's vocabulary of obligation — know what exists before acting (2:31–32), verify before you act (49:6), pursue not that of which you have no knowledge (17:36), and hold what you can damage as a trust (33:72).

@@ -155,7 +157,7 @@

2 The root cause, formally

FacultyWhy it is structurally absentFollows from Memory (across sessions)By P1, nothing survives a turn but the token string; by P3, the string is bounded and lost at session end. There is no addressable store that outlives x.P1, P3 -Learning (from outcomes)By P2, no inference‑time event writes to θ. In‑context “learning” is real optimization14,15 but lives only inside the current x and vanishes with it (P1) — a simulation of learning, not learning.P2, P1 +Learning (from outcomes)By P2, no inference‑time event writes to θ. In‑context “learning” is real optimization14,15 but lives only inside the current x and vanishes with it (P1) — a simulation of learning, not learning. [corrected 2026-09-26]P2, P1 Imagination (simulate before acting)Equation (1) maps tokens to tokens. There is no separate forward model of “what happens to the world (or codebase) if I take action a” distinct from emitting more tokens; the model cannot roll out and score a hypothetical it does not also have to narrate.P1 Self‑correctionAny “check” the model runs is another evaluation of the same fθ with the same blind spots. There is no independent verifier inside Equation (1); the literature confirms intrinsic self‑correction is unreliable without an external signal.21P2 Impact‑awarenessBy P3, the model sees only the tokens in x. A million‑line repository does not fit; therefore it cannot know, unaided, what elsewhere depends on the symbol it is about to change.P3 @@ -165,6 +167,7 @@

2 The root cause, formally

Why prompting and tools do not close the gap

A better prompt changes x. More tools (skills, MCP servers, function calls) let the agent fetch new x or emit richer y. Both operate inside Equation (1) and leave P1–P3 untouched: the composed system is still a stateless map with frozen weights and a bounded window. A tool call retrieves a document into context, but nothing decides what was worth keeping, consolidates it, or updates the agent's priors for next time. The deficits are properties of the loop shape — open, memoryless, one‑directional — not of the model's knowledge. To remove them you must change the shape of the loop, which is precisely what an external substrate can do while θ stays frozen.

+

[corrected 2026-09-26] This callout overstates its case. Prompting and tools do change behaviour: examples, retrieved facts, feedback and extra computation placed in the context adapt a frozen model with no gradient update (Brown et al., 2020). What they do not supply by themselves is four guarantees: durable state across independent invocations, context beyond the window, automatic parameter update from outcomes, and reliable self‑verification without external evidence. The substrate is one tested way of supplying persistence and verification around the model, not the only logically possible architecture.

This reframing is the paper's pivot. If the deficits came from the loop shape, then the remedy is to re‑wrap the loop: keep fθ exactly as it is, and surround it with state and update so that the composite system is no longer memoryless, no longer open, and no longer blind beyond W. Figure 1 states the whole thesis in one picture.

@@ -246,6 +249,8 @@

4.1 What the strongest evidence confirms

has no calibrated sense of its own uncertainty (mechanism M2 below), and it is why “the model felt confident” is not evidence of anything.

+

[corrected 2026-09-26] Scope: this is one randomized trial of 16 experienced developers on 246 tasks in repositories they knew, with early‑2025 tools. It is evidence about that setting, not a universal 2026 productivity coefficient in either direction. METR’s February 2026 update (metr.org/blog/2026-02-24-uplift-update) explains why selection effects complicate newer estimates.

+

The trend evidence is equally well‑sourced. Stack Overflow's 2025 survey of more than 49,000 developers records trust in AI accuracy falling from 40 % to 29 % even as adoption rose to 84 %, with the top‑ranked frustration — cited by 66 % — being code @@ -312,38 +317,51 @@

5 The Qur'anic epistemic lens

How this lens is used — and how it is not

The Qur'an is used here as a framing lens and ethics source, never as technical authority for an engineering claim. No verse is cited to prove that an algorithm works or that a data structure is correct — those claims stand on their engineering merits alone (§3, §8). What the lens supplies is threefold: (1) a precise vocabulary of obligation for what an agent that acts on real systems owes — to truthfulness, to verification, to stewardship; (2) a hierarchy of knowledge (‘ilm → fahm → ḥikma: knowledge → understanding → wisdom) that motivates a layered memory architecture rather than a flat vector store; and (3) ethical constraints on autonomy that translate into concrete safeguards. Where a mapping is marked load‑bearing, the concept motivates a specific design decision (e.g. a mandatory, not optional, verification gate); where marked metaphor, it is illustrative. Canonical text below is presented directly and attributed; it is not paraphrased. Arabic and translations were retrieved from quran.ai; the full 14‑row mapping table is in the appendix.

+

[corrected 2026-09-26] How to read each card. Four layers are kept apart and labelled: the Arabic source text (clean Uthmani script; an ellipsis marks an abridgement, and the full verse is in quranic-lens/quran_lens.md); the English translation (M.A.S. Abdel Haleem); tafsir (classical commentary, Ibn Kathir), which these cards do not quote — the grounding line only records that it was consulted, and the companion file quotes it under its own label; and the author’s design analogy, which is the author’s engineering reading and neither a translation nor a commentary.

The lens earns its place because the deepest failure modes of an autonomous coding agent are not computational but epistemic and ethical: acting without knowing, trusting a report without checking it, and treating a granted capability as license. The Qur'anic vocabulary names these with unusual precision, and three verses in particular map so directly onto architectural decisions that they shaped the design rather than decorating it.

17:36 — lā taqfu · the root of impact‑awareness  [load‑bearing]
+
Arabic source text
وَلَا تَقْفُ مَا لَيْسَ لَكَ بِهِ عِلْمٌ ۚ إِنَّ السَّمْعَ وَالْبَصَرَ وَالْفُؤَادَ كُلُّ أُولَٰئِكَ كَانَ عَنْهُ مَسْئُولًا
+
English translation — M.A.S. Abdel Haleem
“Do not follow blindly what you do not know to be true: ears, eyes, and heart, you will be questioned about all these.”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. Before any mutation (write, delete, refactor), a pre‑action gate must establish what the agent actually knows: which entities the change affects, whether the relevant files/tests/dependents were actually read, and whether the predicted outcome rests on evidence rather than a pattern‑matched guess. Actions taken without verified knowledge are blocked, not merely flagged. The verse's own structure — “ears, eyes, heart… questioned about all these” — maps to an audit trail: every channel the agent used (what it read, inferred, assumed) is logged so the decision can be reconstructed and questioned. This is precisely the impact oracle of §8.
Grounded with quran.ai: fetch_translation(17:36, en-abdel-haleem); fetch_tafsir(17:36, en-ibn-kathir)
49:6 — tabayyun · the verification gate  [load‑bearing]
+
Arabic source text
يَا أَيُّهَا الَّذِينَ آمَنُوا إِن جَاءَكُمْ فَاسِقٌ بِنَبَإٍ فَتَبَيَّنُوا أَن تُصِيبُوا قَوْمًا بِجَهَالَةٍ فَتُصْبِحُوا عَلَىٰ مَا فَعَلْتُمْ نَادِمِينَ
+
English translation — M.A.S. Abdel Haleem
“Believers, if a troublemaker brings you news, check it first, in case you wrong others unwittingly and later regret what you have done.”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. The architecture places a verification step between receiving information (from context, tool output, or its own prior reasoning) and acting on it. The operative term tabayyun demands active investigation, not passive acceptance: before applying a fix based on an error report or its own diagnosis, the agent re‑reads the current file state, confirms the issue still exists, and checks the fix introduces no new breakage detectable by types or tests. The gate is architectural and mandatory — the verse's command is categorical, not conditional on the reporter's trustworthiness — which is exactly the design lesson the self‑correction literature reached empirically21: a real external check, not more self‑prompting.
Grounded with quran.ai: fetch_translation(49:6, en-abdel-haleem); fetch_tafsir(49:6, en-ibn-kathir)
2:31–32 — ta‘līm al‑asmā' · the world‑model  [load‑bearing]
+
Arabic source text
وَعَلَّمَ آدَمَ الْأَسْمَاءَ كُلَّهَا ... قَالُوا سُبْحَانَكَ لَا عِلْمَ لَنَا إِلَّا مَا عَلَّمْتَنَا
+
English translation — M.A.S. Abdel Haleem
“He taught Adam all the names [of things]… They said, ‘May You be glorified! We have knowledge only of what You have taught us.’”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. Knowledge begins with naming — identifying entities and their relations. The agent must hold a structured map of what exists in the codebase (files, functions, classes, dependencies), not a flat listing: knowing that A calls B, that module X depends on Y, that test T covers class C. The angels' admission — “we have knowledge only of what You have taught us” — is a startlingly exact description of the LLM's own situation under P3: it knows only what is in its window. The external world‑model supplies the “names” of entities that exceed context capacity.
Grounded with quran.ai: fetch_translation(2:31-32, en-abdel-haleem)
33:72 — al‑amāna · stewardship & bounded autonomy  [load‑bearing]
+
Arabic source text
إِنَّا عَرَضْنَا الْأَمَانَةَ عَلَى السَّمَاوَاتِ وَالْأَرْضِ وَالْجِبَالِ فَأَبَيْنَ أَن يَحْمِلْنَهَا ... وَحَمَلَهَا الْإِنسَانُ ۖ إِنَّهُ كَانَ ظَلُومًا جَهُولًا
+
English translation — M.A.S. Abdel Haleem
“We offered the Trust to the heavens, the earth, and the mountains, yet they refused to undertake it and were afraid of it; mankind undertook it — they have always been inept and foolish.”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. An agent that can modify a codebase bears an amāna — accepted responsibility for something it can damage. The verse's structure is the design: the heavens declined the trust, recognizing its weight; the human bore it and is called ẓalūman jahūlā (given to wrong and ignorance). So the architecture (a) operates under least privilege — a granted capability is never blanket permission; and (b) assumes the agent will err and builds in reversibility (sandboxing, staged commits, rollback), audit, and scope bounds as structural safeguards. The trust is not “the agent is trustworthy”; it is “the agent has accepted accountability for a domain it can harm, and the architecture must respect that weight.” This is the governance boundary enclosing the entire system in Figure 2.
Grounded with quran.ai: fetch_translation(33:72, en-abdel-haleem); fetch_tafsir(33:72, en-ibn-kathir)
@@ -352,6 +370,8 @@

5 The Qur'anic epistemic lens

The remarkable thing is not that these mappings are poetic; it is that they are operational. “Verify before acting” is not a sentiment here — it is a mandatory gate in the action pipeline. “Know the names of things” is not a metaphor — it is a dependency graph. The lens told us which safeguards are non‑negotiable; the engineering told us how to build them.

+

[corrected 2026-09-26] The operational link is narrower than that sentence suggests. The lens motivates one discipline — do not assert without evidence — which the repository implements as a claim/status registry (docs/status/claims.json), verifier events and visible uncertainty. It establishes no algorithm’s correctness, catch rate or uniqueness: every technical guarantee still needs code‑level assumptions, tests or measurements, and a deterministic implementation does not make a semantic detector’s catch rate approach 1. No theological adjudication is attempted or implied.

+

6 Six mechanisms the frozen loop also lacks

The five faculties of the first edition answer “what cognitive capabilities does a stateless model @@ -754,6 +774,7 @@

11 Genuinely new vs. reinvented

In one line: the components are largely borrowed; the loop shape, the validity anchoring, and the coding‑agent target are the contribution. That is a defensible and useful kind of novelty — it is what turns five scattered literatures into one buildable architecture.

+

[corrected 2026-09-26] Prior art for the combination itself needs naming too: CoALA (Sumers et al., 2023, arXiv:2309.02427) already organises language agents into modular memory, action and decision procedures, and Reflexion18 improves agents through linguistic feedback and an episodic memory buffer without weight updates. The defensible claim is a portable implementation of evidence‑weighted coding‑agent memory and checks, with empirical evaluation of trust failure modes; the “novel” rows above stand only where an explicit comparison with that prior art survives.

12 Limitations & threats to validity

    @@ -765,7 +786,7 @@

    12 Limitations & threats to validity

13 Conclusion

-

The faculties a coding agent seems to lack — memory, learning, imagination, self‑correction, impact‑awareness — are not deficiencies of knowledge that scale will cure. They are structural consequences of what a frozen transformer is: a stateless map with fixed weights and a bounded window (Eq. 1, P1–P3). Because they follow from the shape of the loop, they cannot be prompted or tooled away; they can only be removed by re‑wrapping the loop into a closed, stateful cycle over persistent stores, with the model left frozen inside it (Eq. 2, Fig. 1–2). We specified that substrate faculty by faculty, said honestly which parts are open research and which are engineering, and — for the one faculty that is buildable today — shipped a running impact oracle that, on a package we built, missed no affected file where the strategies a context‑bounded agent actually uses missed up to half. On real repositories it missed almost everything (recall 0.022), which is the refutation’s subject. [refuted — see Corrections] The Qur'anic lens gave the work its spine of obligation: know what exists before you act, verify what you are told, and hold what you can damage as a trust. Those are not just good engineering defaults; here they are the architecture. The next step is to build the memory and learning layers against the same discipline — anchored to what can be verified, not to what the model says of itself — and to evaluate the whole loop on real repositories with real histories.

+

The faculties a coding agent seems to lack — memory, learning, imagination, self‑correction, impact‑awareness — are not deficiencies of knowledge that scale will cure. They are structural consequences of what a frozen transformer is: a stateless map with fixed weights and a bounded window (Eq. 1, P1–P3). Because they follow from the shape of the loop, they cannot be prompted or tooled away; they can only be removed by re‑wrapping the loop into a closed, stateful cycle over persistent stores, with the model left frozen inside it (Eq. 2, Fig. 1–2). [corrected 2026-09-26 — see Corrections] We specified that substrate faculty by faculty, said honestly which parts are open research and which are engineering, and — for the one faculty that is buildable today — shipped a running impact oracle that, on a package we built, missed no affected file where the strategies a context‑bounded agent actually uses missed up to half. On real repositories it missed almost everything (recall 0.022), which is the refutation’s subject. [refuted — see Corrections] The Qur'anic lens gave the work its spine of obligation: know what exists before you act, verify what you are told, and hold what you can damage as a trust. Those are not just good engineering defaults; here they are the architecture. The next step is to build the memory and learning layers against the same discipline — anchored to what can be verified, not to what the model says of itself — and to evaluate the whole loop on real repositories with real histories.


Corrections (2026-09-21)

@@ -779,6 +800,16 @@

Corrections (2026-09-21)

  • The faculty table (§2) is unchanged here, and is now the canonical one. The formal synthesis had a different “follows from” column in all five rows; it has been brought into line with this table, which argues each row and matches what forgekit’s bindings address.
  • +
    +

    Corrections (2026-09-26)

    +

    A second external deep review of the forgekit repository (2026-09-26, pinned at commit d2abfa69fb77531199ffc67c5c076b524af69040) asked what this edition’s argument actually establishes. The changes are marked in place with [corrected 2026-09-26]; the argument itself is not rewritten, and each item quotes the wording it qualifies. The PDF edition predates these corrections as well.

    +
      +
    1. “Cannot learn, imagine or self‑correct” is broader than the argument supports (abstract, §2, §13). The edition says the model “cannot learn from outcomes, cannot imagine the consequences of an action before taking it, cannot reliably correct itself”, that “neither better prompting nor additional tools (skills, MCP servers) remove these deficits”, that in‑context learning is “a simulation of learning, not learning”, and that the faculties “cannot be prompted or tooled away”. Frozen parameters exclude weight updates during use. They do not exclude changed behaviour from examples, retrieved facts, feedback or additional computation in the current context: GPT‑3’s few‑shot evaluation measured exactly that adaptation through text, with no gradient updates (Brown et al., 2020, “Language Models are Few-Shot Learners”, arXiv:2005.14165). What a bare frozen model lacks is four guarantees: no built‑in durable state across independent invocations; a bounded context; no automatic parameter update from outcomes; and unreliable self‑verification without external evidence21. The substrate is one tested way of supplying persistence and verification around the model, not the only logically possible architecture.
    2. +
    3. Prior art and what is (and is not) claimed (§11). External memory, feedback‑driven improvement and structured agent control have clear prior art. CoALA (Sumers et al., 2023, arXiv:2309.02427) organises language agents into modular memory, action and decision procedures; Reflexion18 (Shinn et al., 2023, arXiv:2303.11366) improves agents with linguistic feedback and an episodic memory buffer and no weight updates. The defensible framing is a portable implementation of evidence‑weighted coding‑agent memory and checks, with empirical evaluation of trust failure modes. The “novel composition” and “novel framing” rows of §11 are claims about this combination for coding agents, kept only where an explicit comparison with that prior art survives, and the five faculties are a decomposition, not a proof that these five are necessary.
    4. +
    5. The Qur’anic lens: text, translation, commentary and analogy are now labelled (§5, lens appendix). Each verse card now labels the Arabic source text, the English translation and the author’s design analogy separately, and a legend says where tafsir appears. The appendix column headed “Retrieved gloss” mixed retrieved translation (verse rows) with the author’s own glosses (concept rows), and its “Design principle” column is now headed as the author’s analogy. No Arabic text or translation was altered. The lens motivates the discipline do not assert without evidence, which the repository implements as a claim/status registry, verifier events and visible uncertainty; it establishes no algorithm’s correctness, catch rate or uniqueness, and a deterministic implementation does not imply that a semantic detector’s catch rate approaches 1. No theological adjudication is attempted.
    6. +
    7. METR’s 19% slowdown is scoped (§4.1, evidence appendix C1). The edition calls it “the empirical heart of this paper”. It is a randomized trial of 16 experienced open‑source developers on 246 tasks in repositories they knew, with early‑2025 tools: evidence about that setting, not a universal 2026 productivity coefficient in either direction. METR’s February 2026 update (metr.org/blog/2026-02-24-uplift-update) explains why selection effects complicate newer estimates; it is not evidence that AI now speeds developers up. The evidence map (evidence/evidence_map.md) now grades bibliographic verification, claim support, study design, independent replication and transfer scope separately.
    8. +
    +

    References

      @@ -826,7 +857,7 @@

      Appendix Evidence map — the twelve statist IDClaimPrimary sourceSupportsStatus C1 -Experienced open-source developers were 19% SLOWER with AI while believing they were ~20% faster (also forecast 24% speedup beforehand). +Experienced open-source developers were 19% SLOWER with AI while believing they were ~20% faster (also forecast 24% speedup beforehand). [corrected 2026-09-26] Scope: 16 developers, 246 tasks, early-2025 tools; not a universal coefficient (see Corrections). Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity M2 (assumption/uncertainty - miscalibration), P3/self-correc confirmed @@ -999,9 +1030,9 @@

      Appendix Ecosystem map — faculties &

      Appendix Qur'anic‑lens mapping table

      -

      The complete 14‑row mapping (12 load‑bearing, 2 metaphor). Canonical Arabic and translations retrieved from quran.ai; full text, tafsir references, and design principles in the companion artifact quran_lens.json. Caveat: this table is a design lens, not technical or theological authority — see §5.

      +

      The complete 14‑row mapping (12 load‑bearing, 2 metaphor). Canonical Arabic and translations retrieved from quran.ai; full text, tafsir references, and design principles in the companion artifact quran_lens.json. Caveat: this table is a design lens, not technical or theological authority — see §5. [corrected 2026-09-26] The second column holds the retrieved translation for verse rows and the author’s own gloss for concept rows (it was headed “Retrieved gloss”); the fourth column is the author’s design analogy.

      - + diff --git a/research/cognitive-substrate/evidence/evidence_map.json b/research/cognitive-substrate/evidence/evidence_map.json index bfdbeeb3..40d1f607 100644 --- a/research/cognitive-substrate/evidence/evidence_map.json +++ b/research/cognitive-substrate/evidence/evidence_map.json @@ -13,7 +13,17 @@ "number_as_primary_states": "16 experienced developers, 246 tasks in mature repos (avg 22k+ stars, 1M+ lines); AI use INCREASED completion time by 19%; pre-task forecast was 24% speedup; post-task self-estimate was 20% speedup.", "number_in_field_report": "19% slowdown for experienced devs; matches primary source exactly.", "status": "confirmed", - "note": "Directly reachable on arXiv and METR's own site; numbers match field report exactly. Caveat directly stated by METR: small sample (16 devs), specific to mature/familiar open-source repos, and AI-averse developers increasingly decline to participate (self-selection risk noted by METR itself)." + "note": "Directly reachable on arXiv and METR's own site; numbers match field report exactly. Caveat directly stated by METR: small sample (16 devs), specific to mature/familiar open-source repos, and AI-averse developers increasingly decline to participate (self-selection risk noted by METR itself).", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (arXiv:2507.09089; METR site)", + "claim_support": "supports 19% slower for its own population; does not support a universal or 2026 coefficient in either direction", + "study_design": "randomized controlled trial: 16 experienced developers, 246 tasks", + "independent_replication": "none recorded here; METR's own 2026-02-24 update is by the same group and METR calls it weak evidence because of selection effects", + "transfer_scope": "experienced developers on mature repositories they knew, early-2025 tools", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + }, + "correction_2026_09_26": "Scope: a randomized trial of 16 experienced open-source developers on 246 tasks with early-2025 tools, not a universal 2026 productivity coefficient in either direction. METR's February 2026 update (https://metr.org/blog/2026-02-24-uplift-update/) explains why selection effects complicate newer estimates; do not cite it as evidence that AI now speeds developers up." }, { "claim_id": "C2_SO2025_trust", @@ -29,7 +39,16 @@ "number_as_primary_states": "Trust in AI accuracy fell from 40% (prior years) to 29% (2025); positive favorability fell from 72% to 60%; 84% use or plan to use AI tools (up from 76%); 46% actively distrust AI accuracy vs 33% trust it, only 3% 'highly trust'; 66% cite 'almost right, but not quite' as the #1 frustration, which 'often leads to' the #2 frustration, debugging being more time-consuming (45%).", "number_in_field_report": "Matches primary source essentially exactly on all four sub-figures.", "status": "confirmed", - "note": "Reached directly on Stack Overflow's own survey site and company blog. Self-reported survey; Stack Overflow's own methodology notes flag respondent self-selection bias (recruited via Stack Overflow's own channels, so more AI-engaged/skeptical developers may be over-represented)." + "note": "Reached directly on Stack Overflow's own survey site and company blog. Self-reported survey; Stack Overflow's own methodology notes flag respondent self-selection bias (recruited via Stack Overflow's own channels, so more AI-engaged/skeptical developers may be over-represented).", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (official survey site and blog)", + "claim_support": "supports the trust and favourability figures as survey responses", + "study_design": "self-reported survey, 49,000+ respondents; self-selection noted by Stack Overflow", + "independent_replication": "none recorded here", + "transfer_scope": "Stack Overflow survey respondents, 2025", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C3_Veracode_vuln", @@ -45,7 +64,16 @@ "number_as_primary_states": "Veracode's own report/blog states only the 45% figure (code samples across 100+ LLMs, Java/JS/Python/C# introducing OWASP Top-10 flaws; Java worst at ~72%; XSS failure 86%). Veracode's own public materials located in this search do NOT state a '2.74x' multiplier anywhere.", "number_in_field_report": "Field report attributes BOTH '45%' and '2.74x more vulnerabilities than human-written code' to Veracode, then separately says the 2.74x figure was 'independently corroborated by CodeRabbit's December 2025 analysis of 470 real-world PRs (2.74x more security vulnerabilities...)'.", "status": "vendor-only", - "note": "The 45% figure is directly traceable to Veracode's own report (vendor self-reported, not independently reproduced). The '2.74x' figure could NOT be located in Veracode's own primary materials during this search - it appears only in secondary/derivative blog posts (e.g. softwareseni.com) that attribute it to Veracode, while the field report itself sources the same 2.74x number to a DIFFERENT study (CodeRabbit's PR analysis). This looks like a citation conflation between two separate vendor studies that happen to share a number. Treat the 2.74x figure as unverified/possibly misattributed pending direct access to Veracode's full PDF report." + "note": "The 45% figure is directly traceable to Veracode's own report (vendor self-reported, not independently reproduced). The '2.74x' figure could NOT be located in Veracode's own primary materials during this search - it appears only in secondary/derivative blog posts (e.g. softwareseni.com) that attribute it to Veracode, while the field report itself sources the same 2.74x number to a DIFFERENT study (CodeRabbit's PR analysis). This looks like a citation conflation between two separate vendor studies that happen to share a number. Treat the 2.74x figure as unverified/possibly misattributed pending direct access to Veracode's full PDF report.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "45% confirmed in the vendor report; '2.74x' not found in Veracode's materials", + "claim_support": "supports 45% only; '2.74x' is unsupported (conflated with a separate study)", + "study_design": "vendor benchmark of LLM-generated code samples", + "independent_replication": "none recorded here", + "transfer_scope": "the LLMs, languages and prompts Veracode tested in 2025", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C4_GitClear_duplication", @@ -61,7 +89,16 @@ "number_as_primary_states": "GitClear's own materials are internally inconsistent on the multiplier. The report's own page TITLE reads 'AI Copilot Code Quality: 2025 Data Suggests 4x Growth in Code Clones' (gitclear.com), but the body text of the same report and GitClear's press-mentions page both state duplicated code blocks (5+ lines) 'rose eightfold' / 'increased eightfold' during 2024 (211M changed lines, 2020-2024, Google/Microsoft/Meta/enterprise repos). Secondary summaries split roughly evenly between citing '4x' and '8x'. Underlying non-disputed figures: copy-pasted lines rose from 8.3% (2020) to 12.3% (2024); moved/refactored lines fell from ~24-25% to <10%; 2024 was the first year copy-paste exceeded moved lines. The 4x-vs-8x gap could not be resolved from available pages - likely reflects two different metrics (duplicated-block frequency vs. some other clone measure) reported inconsistently across GitClear's own title/body/press materials.", "number_in_field_report": "Field report states '8x rise in duplicated code blocks' - this matches GitClear's report BODY and press-mentions page, but GitClear's own page TITLE says '4x Growth in Code Clones,' an internal inconsistency the field report does not surface.", "status": "vendor-only", - "note": "GitClear is a code-analytics vendor; findings are corroborated by many independent tech-press writeups summarizing the same underlying dataset, but no independent third party has re-run the analysis separately. Correlational, not causal. IMPORTANT ADDITIONAL CAVEAT: GitClear's own materials are internally inconsistent - the report's page title cites '4x' growth in code clones while the body and press page cite an '8x' rise in duplicated blocks. The paper should either cite the specific metric name (duplicated-block frequency, 8x per body text) rather than a bare multiplier, or note both figures and the discrepancy explicitly rather than asserting '8x' as settled." + "note": "GitClear is a code-analytics vendor; findings are corroborated by many independent tech-press writeups summarizing the same underlying dataset, but no independent third party has re-run the analysis separately. Correlational, not causal. IMPORTANT ADDITIONAL CAVEAT: GitClear's own materials are internally inconsistent - the report's page title cites '4x' growth in code clones while the body and press page cite an '8x' rise in duplicated blocks. The paper should either cite the specific metric name (duplicated-block frequency, 8x per body text) rather than a bare multiplier, or note both figures and the discrepancy explicitly rather than asserting '8x' as settled.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "vendor report found; internally inconsistent (title 4x, body 8x)", + "claim_support": "partial: the direction is supported, the multiplier is not settled", + "study_design": "vendor code-analytics telemetry, correlational", + "independent_replication": "none independent (press summaries re-report the same dataset)", + "transfer_scope": "GitClear's analysed repositories, 2020-2024", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C5_DORA_amplifier", @@ -77,7 +114,16 @@ "number_as_primary_states": "Nearly 5,000 professionals surveyed (June 13-July 21, 2025) plus 100+ hours interviews; 90% AI adoption (14pp increase from 2024); AI's primary role is 'that of an amplifier... magnifying the strengths of high-performing organisations and the dysfunctions of struggling ones'; in 2025 AI's relationship to delivery throughput reversed to positive vs 2024, but AI continues to increase delivery instability; ~30% report little/no trust in AI-generated code.", "number_in_field_report": "Matches primary source directly ('AI is an amplifier'; high adoption; throughput/stability tension).", "status": "confirmed", - "note": "DORA is a Google-run but methodologically transparent, widely-cited industry research program (not a single vendor's self-promotional study); full methodology, sample size and survey window are published. Best-supported of the twelve claims alongside METR." + "note": "DORA is a Google-run but methodologically transparent, widely-cited industry research program (not a single vendor's self-promotional study); full methodology, sample size and survey window are published. Best-supported of the twelve claims alongside METR.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (official DORA report)", + "claim_support": "supports the 'amplifier' framing and the survey figures", + "study_design": "survey of ~5,000 professionals plus interviews; published methodology", + "independent_replication": "none recorded here", + "transfer_scope": "DORA survey respondents, 2025", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C6_SWEbench_retirement", @@ -93,7 +139,16 @@ "number_as_primary_states": "OpenAI audited 138 problems (27.6% subset of the 500-task set) that its o3 model could not reliably solve across 64 runs; found at least 59.4% of THOSE audited problems had flawed test cases/descriptions (35.5% narrow tests, 18.8% wide tests, 5.1% other); also found all tested frontier models could reproduce gold-patch solutions from training memory, indicating contamination. The specific 80.9%-Verified-vs-45.9%-Pro pairing (Claude Opus 4.5) was NOT found stated in OpenAI's own blog; it is reported by third-party benchmark aggregators (e.g. Scale AI SEAL leaderboard, BenchLM.ai, cited via codeant.ai) as of April 2026.", "number_in_field_report": "Field report's phrasing ('59.4% of audited problems had flawed test cases') is accurate to primary source. The 80.9%/45.9% pairing is directionally correct (large real gap exists) but its precise sourcing is a third-party leaderboard snapshot, not OpenAI's own blog post.", "status": "confirmed", - "note": "The 59.4%-of-audited-problems figure is directly confirmed on OpenAI's own site - a strong, well-documented primary source. The specific 80.9/45.9 percentage pair is a real, traceable leaderboard snapshot (Scale AI SEAL/BenchLM) but should be cited as such, not as OpenAI's own number, and will drift as models are re-benchmarked." + "note": "The 59.4%-of-audited-problems figure is directly confirmed on OpenAI's own site - a strong, well-documented primary source. The specific 80.9/45.9 percentage pair is a real, traceable leaderboard snapshot (Scale AI SEAL/BenchLM) but should be cited as such, not as OpenAI's own number, and will drift as models are re-benchmarked.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (OpenAI blog)", + "claim_support": "supports flawed tests in at least 59.4% of a 138-problem audited hard subset, not of all 500 tasks; the 80.9%/45.9% pairing is a third-party leaderboard snapshot", + "study_design": "vendor audit of a benchmark subset, plus contamination probes", + "independent_replication": "none recorded here", + "transfer_scope": "SWE-bench Verified's audited hard subset and the frontier models OpenAI probed", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C7_Faros_PRreview", @@ -109,7 +164,16 @@ "number_as_primary_states": "Two years of telemetry from 22,000 developers / 4,000+ teams, comparing each org's lowest- vs highest-AI-adoption quarters: median time in PR review +441.5% (average time in review +199.6%, first-review wait +156.6%); incidents-to-PR ratio +242.7%; bugs per developer +54%; 31.3% more PRs merged with no review at all; code churn +861%.", "number_in_field_report": "Matches primary source numbers exactly.", "status": "vendor-only", - "note": "Faros AI is an engineering-intelligence vendor whose commercial product monitors exactly these metrics; the report itself and third-party coverage (ADTmag) note these are cross-sectional correlations across the vendor's own customer telemetry, not a controlled study, and 2025-vs-2026 report editions are independent cross-sections rather than a longitudinal panel." + "note": "Faros AI is an engineering-intelligence vendor whose commercial product monitors exactly these metrics; the report itself and third-party coverage (ADTmag) note these are cross-sectional correlations across the vendor's own customer telemetry, not a controlled study, and 2025-vs-2026 report editions are independent cross-sections rather than a longitudinal panel.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed in the vendor report", + "claim_support": "supports the reported deltas as vendor telemetry", + "study_design": "cross-sectional vendor telemetry (22,000 developers), not a controlled study", + "independent_replication": "none recorded here", + "transfer_scope": "Faros AI's customer organisations", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C8_Anthropic_comprehension", @@ -125,7 +189,16 @@ "number_as_primary_states": "(a) Shen & Tamkin: developers who used AI to learn a new async-programming library completed tasks but scored measurably worse on a post-task comprehension test ('17% lower' per a secondary citation in an arXiv survey paper - I could not independently pull Shen & Tamkin's own abstract/number in this search, only a citing paper's paraphrase). (b) The 400,000-session Anthropic study found users make ~70% of planning decisions and Claude makes ~80% of execution decisions; occupation-based success rates were similar across professions (~26-34%); it does NOT report a comprehension-score deficit.", "number_in_field_report": "Field report merges these into one sentence ('Anthropic's own research (~400,000 Claude Code sessions) found... 17% lower on comprehension'), incorrectly attributing the comprehension finding to the session-count study.", "status": "unverifiable", - "note": "This is a citation-conflation error carried over from the field report (or its own sources). The 400K-session study is real and directly confirmed, but does not contain a comprehension-deficit finding. The '17% lower comprehension' figure traces to a separate, distinct Shen & Tamkin paper that this search could not directly retrieve/confirm in primary form (only via a third paper's citation of it). RECOMMENDATION: if the white paper wants to use the comprehension-deficit claim, cite Shen & Tamkin (2026) directly and verify the 17% figure against their own abstract/paper before use; do not attribute it to the 400K-session study." + "note": "This is a citation-conflation error carried over from the field report (or its own sources). The 400K-session study is real and directly confirmed, but does not contain a comprehension-deficit finding. The '17% lower comprehension' figure traces to a separate, distinct Shen & Tamkin paper that this search could not directly retrieve/confirm in primary form (only via a third paper's citation of it). RECOMMENDATION: if the white paper wants to use the comprehension-deficit claim, cite Shen & Tamkin (2026) directly and verify the 17% figure against their own abstract/paper before use; do not attribute it to the 400K-session study.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "unverifiable as attributed (two studies conflated)", + "claim_support": "not supported: the 400,000-session study does not measure comprehension", + "study_design": "not assessed (primary number not retrieved)", + "independent_replication": "not assessed", + "transfer_scope": "not assessed", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C9_Sonar_verification_gap", @@ -141,7 +214,16 @@ "number_as_primary_states": "Survey of 1,100+ (some sources say 1,149) professional developers, January 2026: 96% do not fully trust AI-generated code is functionally correct; only 48% always check AI-assisted code before committing; AI accounts for 42% of committed code (projected 65% by 2027); 38% say reviewing AI code takes more effort than reviewing human code; the term 'verification debt' is attributed to AWS CTO Werner Vogels.", "number_in_field_report": "Matches primary source exactly.", "status": "vendor-only", - "note": "Sonar is a code-quality/verification tooling vendor with a direct commercial interest in this narrative; numbers are self-reported survey data, not independently replicated, though the survey size and methodology are transparently disclosed in the primary PDF." + "note": "Sonar is a code-quality/verification tooling vendor with a direct commercial interest in this narrative; numbers are self-reported survey data, not independently replicated, though the survey size and methodology are transparently disclosed in the primary PDF.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed in the vendor survey", + "claim_support": "supports the 96% / 48% figures as survey responses", + "study_design": "vendor survey, 1,100+ developers", + "independent_replication": "none recorded here", + "transfer_scope": "Sonar survey respondents, January 2026", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C10_JetBrains_manual_correction", @@ -157,7 +239,16 @@ "number_as_primary_states": "JetBrains' own 2025 report (24,534 developers) confirms 85% regularly use AI tools and 62% rely on at least one AI coding assistant, but this search could not locate any statement of a '77% manually correct AI output for conventions every session' figure anywhere in JetBrains' own materials, blog posts, or press coverage of the 2025 or 2026 editions.", "number_in_field_report": "77% manually correct for conventions every session (attributed to JetBrains 2025).", "status": "unverifiable", - "note": "Could not confirm this specific statistic in JetBrains' own primary materials despite multiple targeted searches of the official report, its AI-specific subpage, and secondary coverage. It may be a misremembered/misattributed figure, or drawn from the raw downloadable dataset (500+ questions) rather than the published highlights - the field report should either drop this figure or the paper authors should independently pull it from JetBrains' raw data release before use." + "note": "Could not confirm this specific statistic in JetBrains' own primary materials despite multiple targeted searches of the official report, its AI-specific subpage, and secondary coverage. It may be a misremembered/misattributed figure, or drawn from the raw downloadable dataset (500+ questions) rather than the published highlights - the field report should either drop this figure or the paper authors should independently pull it from JetBrains' raw data release before use.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "not found in JetBrains' materials", + "claim_support": "not supported", + "study_design": "not assessed", + "independent_replication": "not assessed", + "transfer_scope": "not assessed", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C11_MCP_context_bloat", @@ -173,7 +264,16 @@ "number_as_primary_states": "(b) RAG-MCP's own 'MCP stress test' (needle-in-a-haystack-style, N candidate MCP schemas with 1 ground truth) found baseline tool-selection accuracy of 13.62% vs 43.13% for their retrieval-augmented method at scale - i.e. the '43% vs 14%' figures are RAG-MCP's OWN method-vs-baseline comparison on a synthetic stress test, not a general real-world degradation curve as tools accumulate. (a) The 72%/143K-token figure is not from any peer-reviewed or vendor-formal study located in this search; it recurs across multiple blogs as an informal, uncredited individual measurement (one specific developer's personal setup: GitHub + Playwright + IDE MCP servers).", "number_in_field_report": "Field report states these as if they describe general degradation with tool count ('tool-selection accuracy drops from 43% to below 14% as tools accumulate') and cites a 72% context-window consumption figure as an established fact.", "status": "vendor-only", - "note": "The 43.13%-vs-13.62% numbers ARE real and traceable to a genuine arXiv paper (RAG-MCP), but the field report's framing ('as tools accumulate') mischaracterizes what those specific numbers measure (a baseline vs their proposed retrieval method on one synthetic stress test, not a general accumulation curve). The 72%-window figure has no traceable primary/academic source - only recurring, uncredited blog claims. RECOMMENDATION: if used, cite RAG-MCP correctly as 'a stress test showing retrieval-based tool selection outperforms naive selection at scale' rather than a general context-rot statistic, and treat the 72% figure as illustrative anecdote, not a verified finding." + "note": "The 43.13%-vs-13.62% numbers ARE real and traceable to a genuine arXiv paper (RAG-MCP), but the field report's framing ('as tools accumulate') mischaracterizes what those specific numbers measure (a baseline vs their proposed retrieval method on one synthetic stress test, not a general accumulation curve). The 72%-window figure has no traceable primary/academic source - only recurring, uncredited blog claims. RECOMMENDATION: if used, cite RAG-MCP correctly as 'a stress test showing retrieval-based tool selection outperforms naive selection at scale' rather than a general context-rot statistic, and treat the 72% figure as illustrative anecdote, not a verified finding.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "43%/14% traceable to RAG-MCP (arXiv:2505.03275); 72% has no formal source", + "claim_support": "not supported as framed: 43% vs 14% is a method-vs-baseline stress test, not an accumulation curve", + "study_design": "synthetic stress test in a preprint, plus blog anecdotes", + "independent_replication": "none recorded here", + "transfer_scope": "that stress test's setup", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C12_Panickssery_selfpreference", @@ -189,6 +289,15 @@ "number_as_primary_states": "GPT-4 and Llama 2, used as evaluators, have 'non-trivial accuracy' at distinguishing their own outputs from other LLMs' and humans' outputs; a linear correlation is found between self-recognition capability and strength of self-preference bias (LLM evaluators score their own outputs higher while human annotators rate them as equal quality); fine-tuning to improve self-recognition further amplifies self-preference.", "number_in_field_report": "Field report's characterization ('LLMs show self-preference/self-recognition bias when evaluating') matches the paper's core finding faithfully; no specific number is claimed by the field report beyond the qualitative finding.", "status": "confirmed", - "note": "Directly confirmed via the official NeurIPS 2024 proceedings page and the underlying arXiv preprint; a peer-reviewed, widely-cited paper (an NeurIPS 2024 Oral). This is the strongest-quality citation among all twelve (peer-reviewed venue, not industry survey/vendor report)." + "note": "Directly confirmed via the official NeurIPS 2024 proceedings page and the underlying arXiv preprint; a peer-reviewed, widely-cited paper (an NeurIPS 2024 Oral). This is the strongest-quality citation among all twelve (peer-reviewed venue, not industry survey/vendor report).", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (NeurIPS 2024 proceedings)", + "claim_support": "supports the qualitative self-preference finding", + "study_design": "controlled evaluation experiments, peer-reviewed", + "independent_replication": "none recorded here", + "transfer_scope": "the evaluator models studied (GPT-4, Llama 2)", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } } ] \ No newline at end of file diff --git a/research/cognitive-substrate/evidence/evidence_map.md b/research/cognitive-substrate/evidence/evidence_map.md index 901e58d8..76287f5d 100644 --- a/research/cognitive-substrate/evidence/evidence_map.md +++ b/research/cognitive-substrate/evidence/evidence_map.md @@ -19,6 +19,39 @@ | 11 | A standard MCP setup (few servers) can consume ~72% of a 200K-token context window before ... | (b) Qiyao Sun et al. (2025) | ⚠️ vendor-only | See note | | 12 | LLM evaluators recognize and favor their own generations - self-preference bias correlates... | Advances in Neural Information Processing Systems 37 (NeurIPS 2024), Main Conference Track (Oral) (2024) | ✅ confirmed | See note | +## Five kinds of grade, kept separate (added 2026-09-26) + +The `Status` column above records mainly **bibliographic verification** and source independence: +whether a number can be traced to a primary source, and whether the party behind it has a stake. It +is not a grade of whether the source supports the claim as used, of its design, of replication, or of +how far the result travels, so "confirmed" does not mean "generalizes". The table below grades those +dimensions separately, using only what each entry in this map already records. + +| Dimension | Question it answers | +|---|---| +| Bibliographic verification | Does the cited source exist, is it correctly attributed, and does it contain the number? | +| Claim support | Does the source support the claim *as this paper uses it* (population, measure, direction)? | +| Study design | Randomized trial, observational study, survey, vendor telemetry, benchmark audit? | +| Independent replication | Has anyone other than the originator reproduced it? | +| Transfer scope | Which population, date, tooling and setting can the result be carried to? | + +| Claim | Bibliographic verification | Claim support | Study design | Independent replication | Transfer scope | +|---|---|---|---|---|---| +| C1 METR slowdown | confirmed (arXiv:2507.09089; METR site) | supports 19% slower for its own population; does not support a universal or 2026 coefficient in either direction | randomized controlled trial: 16 experienced developers, 246 tasks | none recorded here; METR's own 2026-02-24 update is by the same group and METR calls it weak evidence because of selection effects | experienced developers on mature repositories they knew, early-2025 tools | +| C2 Stack Overflow trust | confirmed (official survey site and blog) | supports the trust and favourability figures as survey responses | self-reported survey, 49,000+ respondents; self-selection noted by Stack Overflow | none recorded here | Stack Overflow survey respondents, 2025 | +| C3 Veracode vulnerabilities | 45% confirmed in the vendor report; '2.74x' not found in Veracode's materials | supports 45% only; '2.74x' is unsupported (conflated with a separate study) | vendor benchmark of LLM-generated code samples | none recorded here | the LLMs, languages and prompts Veracode tested in 2025 | +| C4 GitClear duplication | vendor report found; internally inconsistent (title 4x, body 8x) | partial: the direction is supported, the multiplier is not settled | vendor code-analytics telemetry, correlational | none independent (press summaries re-report the same dataset) | GitClear's analysed repositories, 2020-2024 | +| C5 DORA amplifier | confirmed (official DORA report) | supports the 'amplifier' framing and the survey figures | survey of ~5,000 professionals plus interviews; published methodology | none recorded here | DORA survey respondents, 2025 | +| C6 SWE-bench Verified audit | confirmed (OpenAI blog) | supports flawed tests in at least 59.4% of a 138-problem audited hard subset, not of all 500 tasks; the 80.9%/45.9% pairing is a third-party leaderboard snapshot | vendor audit of a benchmark subset, plus contamination probes | none recorded here | SWE-bench Verified's audited hard subset and the frontier models OpenAI probed | +| C7 Faros PR review | confirmed in the vendor report | supports the reported deltas as vendor telemetry | cross-sectional vendor telemetry (22,000 developers), not a controlled study | none recorded here | Faros AI's customer organisations | +| C8 comprehension (conflated) | unverifiable as attributed (two studies conflated) | not supported: the 400,000-session study does not measure comprehension | not assessed (primary number not retrieved) | not assessed | not assessed | +| C9 Sonar verification gap | confirmed in the vendor survey | supports the 96% / 48% figures as survey responses | vendor survey, 1,100+ developers | none recorded here | Sonar survey respondents, January 2026 | +| C10 JetBrains manual correction | not found in JetBrains' materials | not supported | not assessed | not assessed | not assessed | +| C11 MCP context bloat | 43%/14% traceable to RAG-MCP (arXiv:2505.03275); 72% has no formal source | not supported as framed: 43% vs 14% is a method-vs-baseline stress test, not an accumulation curve | synthetic stress test in a preprint, plus blog anecdotes | none recorded here | that stress test's setup | +| C12 self-preference bias | confirmed (NeurIPS 2024 proceedings) | supports the qualitative self-preference finding | controlled evaluation experiments, peer-reviewed | none recorded here | the evaluator models studied (GPT-4, Llama 2) | + +The same grades are in `evidence_map.json` under each record's `grades` field. + ## Detailed Findings ### 1. C1_METR_slowdown: ✅ confirmed @@ -35,6 +68,8 @@ **Note:** Directly reachable on arXiv and METR's own site; numbers match field report exactly. Caveat directly stated by METR: small sample (16 devs), specific to mature/familiar open-source repos, and AI-averse developers increasingly decline to participate (self-selection risk noted by METR itself). +**Correction (2026-09-26):** scope this wherever C1 appears. It is a study of 16 experienced developers on 246 tasks with early-2025 tooling, not a universal 2026 productivity coefficient in either direction. METR's [February 2026 update](https://metr.org/blog/2026-02-24-uplift-update/) explains why selection effects complicate newer estimates; it is not evidence that AI now speeds developers up either. (This scoping was recommended in `research/formal-synthesis/audits/our_evidence_corrections.json` and had not been applied here.) + ### 2. C2_SO2025_trust: ✅ confirmed **Claim:** Trust in AI accuracy fell from 40% to 29% (or 46% actively distrust vs 33% trust per detailed breakdown); favorability 72%->60%; 66% say AI answers are 'almost right, but not quite'; 45% say debugging AI code is more time-consuming. @@ -193,7 +228,7 @@ **Safe to lean on without hedging (peer-reviewed / official primary source, methodology transparent):** -- METR's 19%-slowdown RCT (C1) — the single best-controlled empirical finding in the set; cite with its own caveats (n=16, mature-repo setting). +- METR's 19%-slowdown RCT (C1) — the single best-controlled empirical finding in the set; cite with its own caveats (n=16, mature-repo setting, early-2025 tools; not a universal 2026 coefficient — see the 2026-09-26 correction under C1). - Panickssery et al. NeurIPS 2024 self-preference bias (C12) — the only genuinely peer-reviewed academic paper among the twelve; strongest citation for the M6 argument that an LLM cannot be its sole verifier. - OpenAI's own retirement of SWE-bench Verified and the 59.4%-of-audited-problems figure (C6) — directly stated on OpenAI's blog. The specific 80.9%/45.9% score pairing, however, should be cited as a third-party leaderboard snapshot (Scale AI SEAL/BenchLM), not as OpenAI's own number, since it will drift release-to-release. - DORA 2025 "AI is an amplifier" finding (C5) — large, transparent, non-vendor-captured methodology (Google Cloud + independent research partners), the most credible of the survey-based claims. diff --git a/research/cognitive-substrate/quranic-lens/quran_lens.json b/research/cognitive-substrate/quranic-lens/quran_lens.json index effd3540..cf7315c9 100644 --- a/research/cognitive-substrate/quranic-lens/quran_lens.json +++ b/research/cognitive-substrate/quranic-lens/quran_lens.json @@ -7,6 +7,14 @@ "Grounded with quran.ai: fetch_tafsir(49:6, en-ibn-kathir)", "Grounded with quran.ai: fetch_tafsir(33:72, en-ibn-kathir)" ], + "field_legend": { + "added": "2026-09-26", + "arabic_or_ref": "verse rows: the source text in clean Uthmani script (retrieved); concept rows: the Arabic term and root", + "retrieved_translation_or_gloss": "verse rows: M.A.S. Abdel Haleem's translation (retrieved); concept rows: the author's own gloss. The two are different kinds of text.", + "concrete_design_principle": "the author's design analogy: not a translation, not tafsir, and not a claim about what the verse means", + "source": "retrieval record; a tafsir named here was consulted, and quran_lens.md quotes it under its own label", + "operational_note": "The lens motivates the 'do not assert without evidence' discipline (claim registry, verifier events, visible uncertainty). It establishes no algorithm's correctness, catch rate or uniqueness, and a deterministic implementation does not imply c -> 1 for a semantic detector. No theological adjudication is attempted." + }, "mappings": [ { "concept_or_verse": "17:36 — lā taqfu (do not pursue without knowledge)", diff --git a/research/cognitive-substrate/quranic-lens/quran_lens.md b/research/cognitive-substrate/quranic-lens/quran_lens.md index 1e8c84b1..2c651b17 100644 --- a/research/cognitive-substrate/quranic-lens/quran_lens.md +++ b/research/cognitive-substrate/quranic-lens/quran_lens.md @@ -4,6 +4,33 @@ The Quran is used here as a FRAMING LENS and ETHICS SOURCE for the cognitive substrate design, never as technical authority for an engineering claim. No verse is cited to prove that a particular algorithm works or that a specific data structure is correct — those claims stand or fall on their engineering merits alone. What the Quranic framing provides is: (1) a vocabulary for naming the agent's epistemic obligations (what it owes to truthfulness, to verification, to stewardship), (2) a hierarchy of knowledge (ʿilm → fahm → ḥikma) that motivates a layered memory architecture rather than a flat one, and (3) ethical constraints on autonomy (amāna, tabayyun) that translate into concrete architectural safeguards. Where a mapping is marked 'metaphor,' the analogy is illustrative — it communicates the design motivation but does not uniquely determine the technical solution. Where a mapping is marked 'load-bearing,' the Quranic concept directly motivates a specific architectural decision (e.g., a mandatory verification gate, not an optional one). Even in load-bearing cases, the engineering justification must be independently defensible — the verse explains *why* we insist on this design choice, not *that* it will work. Discovered patterns in the text describe; they do not legislate. + +## How to read each entry (labels added 2026-09-26) + +Every entry keeps four kinds of text apart, each under its own label. When these labels were +added, no Arabic text and no translation was altered, re-translated or paraphrased; paragraphs +that mixed a tafsir report with the author's reading were split at the sentence boundary. + +| Label | What it is | Whose words | +|---|---|---| +| **Arabic (source text)** / **Arabic (term and root)** | the verse in clean Uthmani script, or the Arabic term a concept entry discusses | the source text, retrieved (see Grounding) | +| **Translation (Abdel Haleem)** | the English translation of the verse | M.A.S. Abdel Haleem, retrieved | +| **Tafsir (Ibn Kathir, as reported)** | a summary of classical commentary and any quotation it carries | Ibn Kathir's tafsir, as reported by the author | +| **Gloss (author's)** | a short explanation of a concept | the author | +| **Design principle — author's analogy** (load-bearing or metaphor) | the engineering reading the author draws | the author: not a translation, not tafsir, and not a claim about what the verse means | + +## What the lens does and does not establish (2026-09-26) + +The lens is motivation, not evidence. Operationally it motivates one discipline — *do not assert +without evidence* — which this repository implements as a machine-readable claim/status registry +([`docs/status/claims.json`](../../../docs/status/claims.json)), verifier events that record what +actually ran, and uncertainty that is shown to the user rather than hidden. It establishes no +algorithm's correctness, catch rate or uniqueness: every technical guarantee still needs code-level +assumptions, tests or measurements. In particular, a deterministic implementation of a check does +not imply that its catch rate on semantic failures approaches 1 (`c → 1`); a deterministic gate is +exact about its proxy signal, not about the miss. No theological adjudication is attempted or +implied. + --- ## Grounding @@ -22,17 +49,19 @@ All verse translations below are retrieved canonical text from the Abdel Haleem ### 17:36 — lā taqfu mā laysa laka bihi ʿilm -**Arabic:** +**Arabic (source text):** > وَلَا تَقْفُ مَا لَيْسَ لَكَ بِهِ عِلْمٌ ۚ إِنَّ السَّمْعَ وَالْبَصَرَ وَالْفُؤَادَ كُلُّ أُولَٰئِكَ كَانَ عَنْهُ مَسْئُولًا **Translation (Abdel Haleem):** > Do not follow blindly what you do not know to be true: ears, eyes, and heart, you will be questioned about all these. -**Design principle (load-bearing):** Before any code mutation (file write, delete, refactor), the agent must run a pre-action verification gate that checks: (1) what entities in the codebase will be affected, (2) whether the agent has sufficient context (has it read the relevant files, tests, and dependents), and (3) whether the predicted outcome is supported by evidence rather than pattern-matched guessing. Actions taken without verified knowledge are blocked, not merely flagged. +**Design principle — author's analogy (load-bearing):** Before any code mutation (file write, delete, refactor), the agent must run a pre-action verification gate that checks: (1) what entities in the codebase will be affected, (2) whether the agent has sufficient context (has it read the relevant files, tests, and dependents), and (3) whether the predicted outcome is supported by evidence rather than pattern-matched guessing. Actions taken without verified knowledge are blocked, not merely flagged. The verse's structure — "ears, eyes, and heart, you will be questioned about all these" — maps to an audit trail: every sensory channel the agent used (what it read, what it inferred, what it assumed) is logged so the decision can be reconstructed and questioned. -Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting without knowledge, citing Qatadah: "Do not say 'I have seen' when you did not see anything, or 'I have heard' when you did not hear anything, or 'I know' when you do not know, for Allah will ask you about all of that." For the agent, the parallel is direct: do not claim a file is safe to modify when you have not read it, do not assert a test passes when you have not run it, and do not say a change is isolated when you have not traced its dependents. +**Tafsir (Ibn Kathir, as reported):** Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting without knowledge, citing Qatadah: "Do not say 'I have seen' when you did not see anything, or 'I have heard' when you did not hear anything, or 'I know' when you do not know, for Allah will ask you about all of that." + +**Design principle — author's analogy (continued):** For the agent, the parallel is direct: do not claim a file is safe to modify when you have not read it, do not assert a test passes when you have not run it, and do not say a change is isolated when you have not traced its dependents. --- @@ -40,49 +69,49 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 49:6 — tabayyun (the verification gate) -**Arabic:** +**Arabic (source text):** > يَا أَيُّهَا الَّذِينَ آمَنُوا إِن جَاءَكُمْ فَاسِقٌ بِنَبَإٍ فَتَبَيَّنُوا أَن تُصِيبُوا قَوْمًا بِجَهَالَةٍ فَتُصْبِحُوا عَلَىٰ مَا فَعَلْتُمْ نَادِمِينَ **Translation (Abdel Haleem):** > Believers, if a troublemaker brings you news, check it first, in case you wrong others unwittingly and later regret what you have done, -**Design principle (load-bearing):** The agent architecture must include a verification gate between receiving information and acting on it. The verse's operative term tabayyun (تَبَيُّنُوا) demands active investigation, not passive acceptance. Before applying a code change based on an error report, a user request, or its own diagnosis, the agent must independently verify the claim — re-read the file, re-run the test, check that the error still exists. This prevents cascading damage from stale context, hallucinated errors, or misunderstood instructions. The gate is architectural (a mandatory step in the action pipeline), not advisory. +**Design principle — author's analogy (load-bearing):** The agent architecture must include a verification gate between receiving information and acting on it. The verse's operative term tabayyun (تَبَيُّنُوا) demands active investigation, not passive acceptance. Before applying a code change based on an error report, a user request, or its own diagnosis, the agent must independently verify the claim — re-read the file, re-run the test, check that the error still exists. This prevents cascading damage from stale context, hallucinated errors, or misunderstood instructions. The gate is architectural (a mandatory step in the action pipeline), not advisory. ### 4:82 — tadabbur as self-consistency checking -**Arabic:** +**Arabic (source text):** > أَفَلَا يَتَدَبَّرُونَ الْقُرْآنَ ۚ وَلَوْ كَانَ مِنْ عِندِ غَيْرِ اللَّهِ لَوَجَدُوا فِيهِ اخْتِلَافًا كَثِيرًا **Translation (Abdel Haleem):** > Will they not think about this Quran? If it had been from anyone other than God, they would have found much inconsistency in it. -**Design principle (load-bearing):** The agent must run self-consistency checks on its own output before committing it. The verse's argument is structural: internal contradiction is evidence of flawed origin. If a planned set of code changes contradicts the agent's own stated reasoning, or if the predicted outcome of an edit conflicts with the test expectations the agent just read, the system should flag the inconsistency and halt. This is the metacognitive controller — a structured reflection pass, not a vague "think again" prompt. +**Design principle — author's analogy (load-bearing):** The agent must run self-consistency checks on its own output before committing it. The verse's argument is structural: internal contradiction is evidence of flawed origin. If a planned set of code changes contradicts the agent's own stated reasoning, or if the predicted outcome of an edit conflicts with the test expectations the agent just read, the system should flag the inconsistency and halt. This is the metacognitive controller — a structured reflection pass, not a vague "think again" prompt. ### 47:24 — tadabbur as deliberate re-examination -**Arabic:** +**Arabic (source text):** > أَفَلَا يَتَدَبَّرُونَ الْقُرْآنَ أَمْ عَلَىٰ قُلُوبٍ أَقْفَالُهَا **Translation (Abdel Haleem):** > Will they not contemplate the Quran? Do they have locks on their hearts? -**Design principle (metaphor):** The "locks on hearts" image maps to a real architectural failure mode: when the agent's context is saturated or its attention is consumed by irrelevant detail, it becomes functionally locked — unable to reconsider its approach. The metacognitive controller must be able to reset the agent's working context, re-examine the problem from a fresh framing, and iterate. This means the reflection loop can propose and evaluate alternative plans — a structured backtracking mechanism. +**Design principle — author's analogy (metaphor):** The "locks on hearts" image maps to a real architectural failure mode: when the agent's context is saturated or its attention is consumed by irrelevant detail, it becomes functionally locked — unable to reconsider its approach. The metacognitive controller must be able to reset the agent's working context, re-examine the problem from a fresh framing, and iterate. This means the reflection loop can propose and evaluate alternative plans — a structured backtracking mechanism. ### CONCEPT: tadabbur (deep, structured reflection) -**Arabic:** تَدَبُّر (root: د-ب-ر, relating to what comes after, consequences) +**Arabic (term and root):** تَدَبُّر (root: د-ب-ر, relating to what comes after, consequences) -**Gloss:** Tadabbur is not casual thought; its root d-b-r relates to "what is behind" or "what follows" — examining the consequences and deeper implications. In Quranic usage (4:82, 47:24), it is the deliberate act of looking beyond the surface to the structure beneath. +**Gloss (author's):** Tadabbur is not casual thought; its root d-b-r relates to "what is behind" or "what follows" — examining the consequences and deeper implications. In Quranic usage (4:82, 47:24), it is the deliberate act of looking beyond the surface to the structure beneath. -**Design principle (load-bearing):** The metacognitive controller is a tadabbur loop: after the agent generates a plan, the controller examines what comes after — what are the downstream consequences of this change? What will break? What assumptions does this rely on? This is not a confidence score but a structured trace-forward through the dependency graph. +**Design principle — author's analogy (load-bearing):** The metacognitive controller is a tadabbur loop: after the agent generates a plan, the controller examines what comes after — what are the downstream consequences of this change? What will break? What assumptions does this rely on? This is not a confidence score but a structured trace-forward through the dependency graph. ### CONCEPT: tabayyun (verification before action) -**Arabic:** تَبَيُّن (root: ب-ي-ن, clarity, making evident) +**Arabic (term and root):** تَبَيُّن (root: ب-ي-ن, clarity, making evident) -**Gloss:** Tabayyun is the act of seeking clarity and verification before acting on received information. In 49:6, it is commanded as a mandatory step between receiving a report and taking action, specifically to prevent harm caused by acting on unverified information. +**Gloss (author's):** Tabayyun is the act of seeking clarity and verification before acting on received information. In 49:6, it is commanded as a mandatory step between receiving a report and taking action, specifically to prevent harm caused by acting on unverified information. -**Design principle (load-bearing):** The verification gate sits between the agent's diagnosis and its action. The gate requires: (1) re-read the actual current state of files to be modified, (2) confirm the error still exists and matches the diagnosis, (3) verify the proposed fix does not introduce new issues. The verse's command is categorical — not conditional on confidence level. +**Design principle — author's analogy (load-bearing):** The verification gate sits between the agent's diagnosis and its action. The gate requires: (1) re-read the actual current state of files to be modified, (2) confirm the error still exists and matches the diagnosis, (3) verify the proposed fix does not introduce new issues. The verse's command is categorical — not conditional on confidence level. --- @@ -90,7 +119,7 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 96:1-5 — iqraʾ / ʿallama bi-l-qalam (Read; taught by the pen) -**Arabic:** +**Arabic (source text):** > اقْرَأْ بِاسْمِ رَبِّكَ الَّذِي خَلَقَ ﴿١﴾ > خَلَقَ الْإِنسَانَ مِنْ عَلَقٍ ﴿٢﴾ > اقْرَأْ وَرَبُّكَ الْأَكْرَمُ ﴿٣﴾ @@ -104,15 +133,15 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting > (96:4) who taught by [means of] the pen, > (96:5) who taught man what he did not know. -**Design principle (load-bearing):** Knowledge must be externalized to survive beyond the moment of computation. The pen (al-qalam) is the instrument of externalization — it transforms ephemeral thought into durable record. Every context window is ephemeral (like unwritten thought), so a persistent memory store (the "pen") must write down what the agent learns, decides, and observes. The architecture requires: (a) a write-back mechanism that captures salient facts into durable storage, (b) a retrieval mechanism that re-loads relevant past experience into the next session's context, (c) a consolidation process that organizes raw experience into structured knowledge. "Taught man what he did not know" — the pen does not just record; it enables access to knowledge beyond unaided capacity. +**Design principle — author's analogy (load-bearing):** Knowledge must be externalized to survive beyond the moment of computation. The pen (al-qalam) is the instrument of externalization — it transforms ephemeral thought into durable record. Every context window is ephemeral (like unwritten thought), so a persistent memory store (the "pen") must write down what the agent learns, decides, and observes. The architecture requires: (a) a write-back mechanism that captures salient facts into durable storage, (b) a retrieval mechanism that re-loads relevant past experience into the next session's context, (c) a consolidation process that organizes raw experience into structured knowledge. "Taught man what he did not know" — the pen does not just record; it enables access to knowledge beyond unaided capacity. ### CONCEPT: ḥifẓ + murājaʿa (preservation + spaced review) -**Arabic:** حِفْظ + مُرَاجَعَة +**Arabic (term and root):** حِفْظ + مُرَاجَعَة -**Gloss:** The classical Quranic memorization discipline: ḥifẓ is initial encoding and faithful preservation; murājaʿa is the regular, spaced revision that prevents decay. Together they form a complete memory system — encoding plus maintenance. +**Gloss (author's):** The classical Quranic memorization discipline: ḥifẓ is initial encoding and faithful preservation; murājaʿa is the regular, spaced revision that prevents decay. Together they form a complete memory system — encoding plus maintenance. -**Design principle (load-bearing):** Memory is not write-once. The agent's persistent store requires a maintenance cycle: periodic review to (a) reinforce high-value patterns that recur, (b) decay or archive entries that have not been accessed or validated, (c) detect and resolve contradictions between old and new experience. The ḥifẓ principle also demands fidelity: what is stored must accurately represent what happened, not a lossy summary that drifts from the original. Concrete mechanism: a background consolidation process that scores memories by recency, frequency, and outcome relevance, and prunes low-scoring entries. +**Design principle — author's analogy (load-bearing):** Memory is not write-once. The agent's persistent store requires a maintenance cycle: periodic review to (a) reinforce high-value patterns that recur, (b) decay or archive entries that have not been accessed or validated, (c) detect and resolve contradictions between old and new experience. The ḥifẓ principle also demands fidelity: what is stored must accurately represent what happened, not a lossy summary that drifts from the original. Concrete mechanism: a background consolidation process that scores memories by recency, frequency, and outcome relevance, and prunes low-scoring entries. --- @@ -120,31 +149,31 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 20:114 — rabbi zidnī ʿilmā (My Lord, increase me in knowledge) -**Arabic:** +**Arabic (source text):** > فَتَعَالَى اللَّهُ الْمَلِكُ الْحَقُّ ۗ وَلَا تَعْجَلْ بِالْقُرْآنِ مِن قَبْلِ أَن يُقْضَىٰ إِلَيْكَ وَحْيُهُ ۖ وَقُل رَّبِّ زِدْنِي عِلْمًا **Translation (Abdel Haleem):** > exalted be God, the one who is truly in control. [Prophet], do not rush to recite before the revelation is fully complete but say, ‘Lord, increase me in knowledge!’ -**Design principle (load-bearing):** The agent's knowledge must be treated as perpetually incomplete, with an explicit mechanism for incremental growth. The prayer "increase me in knowledge" implies knowledge is not a fixed endowment but an ongoing accumulation. The system maintains a learning store (patterns observed, errors encountered, user corrections accepted) that grows across sessions. Each session's outcomes feed back into a persistent experience store that updates the agent's priors for future sessions. This is not fine-tuning (the LLM weights stay frozen); it is an external memory that changes what the agent sees on its next input. +**Design principle — author's analogy (load-bearing):** The agent's knowledge must be treated as perpetually incomplete, with an explicit mechanism for incremental growth. The prayer "increase me in knowledge" implies knowledge is not a fixed endowment but an ongoing accumulation. The system maintains a learning store (patterns observed, errors encountered, user corrections accepted) that grows across sessions. Each session's outcomes feed back into a persistent experience store that updates the agent's priors for future sessions. This is not fine-tuning (the LLM weights stay frozen); it is an external memory that changes what the agent sees on its next input. ### 39:9 — hal yastawī (are those who know equal to those who do not know?) -**Arabic:** +**Arabic (source text):** > أَمَّنْ هُوَ قَانِتٌ آنَاءَ اللَّيْلِ سَاجِدًا وَقَائِمًا يَحْذَرُ الْآخِرَةَ وَيَرْجُو رَحْمَةَ رَبِّهِ ۗ قُلْ هَلْ يَسْتَوِي الَّذِينَ يَعْلَمُونَ وَالَّذِينَ لَا يَعْلَمُونَ ۗ إِنَّمَا يَتَذَكَّرُ أُولُو الْأَلْبَابِ **Translation (Abdel Haleem):** > What about someone who worships devoutly during the night, bowing down, standing in prayer, ever mindful of the life to come, hoping for his Lord’s mercy? Say, ‘How can those who know be equal to those who do not know?’ Only those who have understanding will take heed. -**Design principle (metaphor):** An agent that retains and learns from experience is categorically more capable and more trustworthy than one that does not. The verse establishes that knowledge is not fungible with ignorance; they produce different outcomes. The architecture must distinguish between the agent operating with relevant prior experience loaded (grounded mode) versus from the base model alone (ungrounded mode), and should surface this distinction to the user. +**Design principle — author's analogy (metaphor):** An agent that retains and learns from experience is categorically more capable and more trustworthy than one that does not. The verse establishes that knowledge is not fungible with ignorance; they produce different outcomes. The architecture must distinguish between the agent operating with relevant prior experience loaded (grounded mode) versus from the base model alone (ungrounded mode), and should surface this distinction to the user. ### CONCEPT: ʿilm → fahm → ḥikma (knowledge → understanding → wisdom) -**Arabic:** عِلْم → فَهْم → حِكْمَة +**Arabic (term and root):** عِلْم → فَهْم → حِكْمَة -**Gloss:** A classical epistemological hierarchy: ʿilm is raw knowledge (facts, data); fahm is comprehension (grasping relations, seeing why); ḥikma is wisdom (knowing what to do with understanding — right action at the right time). +**Gloss (author's):** A classical epistemological hierarchy: ʿilm is raw knowledge (facts, data); fahm is comprehension (grasping relations, seeing why); ḥikma is wisdom (knowing what to do with understanding — right action at the right time). -**Design principle (load-bearing):** The agent's memory/learning stack must be layered, not flat. Raw experience logs (ʿilm) are the base layer. A consolidation process extracts patterns and relationships (fahm). A decision-support layer (ḥikma) applies these patterns to new situations. Each layer has different storage, update, and retrieval characteristics. Dumping everything into a flat vector store collapses the hierarchy and loses the distinction between raw fact and actionable understanding. +**Design principle — author's analogy (load-bearing):** The agent's memory/learning stack must be layered, not flat. Raw experience logs (ʿilm) are the base layer. A consolidation process extracts patterns and relationships (fahm). A decision-support layer (ḥikma) applies these patterns to new situations. Each layer has different storage, update, and retrieval characteristics. Dumping everything into a flat vector store collapses the hierarchy and loses the distinction between raw fact and actionable understanding. --- @@ -152,7 +181,7 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 2:31-32 — taʿlīm al-asmāʾ (He taught Adam the names) -**Arabic:** +**Arabic (source text):** > وَعَلَّمَ آدَمَ الْأَسْمَاءَ كُلَّهَا ثُمَّ عَرَضَهُمْ عَلَى الْمَلَائِكَةِ فَقَالَ أَنبِئُونِي بِأَسْمَاءِ هَٰؤُلَاءِ إِن كُنتُمْ صَادِقِينَ ﴿٣١﴾ > قَالُوا سُبْحَانَكَ لَا عِلْمَ لَنَا إِلَّا مَا عَلَّمْتَنَا ۖ إِنَّكَ أَنتَ الْعَلِيمُ الْحَكِيمُ ﴿٣٢﴾ @@ -160,7 +189,7 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting > (2:31) He taught Adam all the names [of things], then He showed them to the angels and said, ‘Tell me the names of these if you truly [think you can].’ > (2:32) They said, ‘May You be glorified! We have knowledge only of what You have taught us. You are the All Knowing and All Wise.’ -**Design principle (load-bearing):** The agent must maintain a structured representation of what exists in the codebase — a graph of files, functions, classes, dependencies, and their relationships. This is not a flat file listing but a semantic map: knowing that function A calls function B, that module X depends on module Y, that test T covers class C. The verse's point is that knowledge begins with naming — identifying entities and their natures. The angels' admission "we have knowledge only of what You have taught us" maps precisely to the LLM's situation: it knows only what is in its context window. The external world-model compensates by providing the "names" (identities and relations) of codebase entities that exceed context capacity. +**Design principle — author's analogy (load-bearing):** The agent must maintain a structured representation of what exists in the codebase — a graph of files, functions, classes, dependencies, and their relationships. This is not a flat file listing but a semantic map: knowing that function A calls function B, that module X depends on module Y, that test T covers class C. The verse's point is that knowledge begins with naming — identifying entities and their natures. The angels' admission "we have knowledge only of what You have taught us" maps precisely to the LLM's situation: it knows only what is in its context window. The external world-model compensates by providing the "names" (identities and relations) of codebase entities that exceed context capacity. --- @@ -168,23 +197,25 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 33:72 — al-amāna (the Trust) -**Arabic:** +**Arabic (source text):** > إِنَّا عَرَضْنَا الْأَمَانَةَ عَلَى السَّمَاوَاتِ وَالْأَرْضِ وَالْجِبَالِ فَأَبَيْنَ أَن يَحْمِلْنَهَا وَأَشْفَقْنَ مِنْهَا وَحَمَلَهَا الْإِنسَانُ ۖ إِنَّهُ كَانَ ظَلُومًا جَهُولًا **Translation (Abdel Haleem):** > We offered the Trust to the heavens, the earth, and the mountains, yet they refused to undertake it and were afraid of it; mankind undertook it- they have always been inept and foolish. -**Design principle (load-bearing):** An agent that can modify a codebase bears a trust (amāna). The verse's structure is crucial: the heavens and earth refused the trust, recognizing its weight; the human bore it and was described as ẓalūman jahūlā (given to wrongdoing and ignorance). The design implication is dual: (a) the agent must operate within explicit bounds of authorization — it may not exceed the scope of what it was asked to do; (b) the architecture must assume the agent will err (jahūl) and build in rollback, sandboxing, and incremental commit as structural safeguards. The trust is not "the agent is trustworthy"; the trust is "the agent has accepted accountability for a domain it can harm, and the architecture must respect that weight." +**Design principle — author's analogy (load-bearing):** An agent that can modify a codebase bears a trust (amāna). The verse's structure is crucial: the heavens and earth refused the trust, recognizing its weight; the human bore it and was described as ẓalūman jahūlā (given to wrongdoing and ignorance). The design implication is dual: (a) the agent must operate within explicit bounds of authorization — it may not exceed the scope of what it was asked to do; (b) the architecture must assume the agent will err (jahūl) and build in rollback, sandboxing, and incremental commit as structural safeguards. The trust is not "the agent is trustworthy"; the trust is "the agent has accepted accountability for a domain it can harm, and the architecture must respect that weight." + +**Tafsir (Ibn Kathir, as reported):** Ibn Kathir's tafsir reports Ibn Abbas identifying the amāna with obedience and accountability: "If you do good, you will be rewarded, and if you do evil, you will be punished." -Ibn Kathir's tafsir reports Ibn Abbas identifying the amāna with obedience and accountability: "If you do good, you will be rewarded, and if you do evil, you will be punished." For the agent, this translates to outcome-linked feedback: the agent's actions must be traceable to outcomes, and those outcomes must feed back into the learning store. +**Design principle — author's analogy (continued):** For the agent, this translates to outcome-linked feedback: the agent's actions must be traceable to outcomes, and those outcomes must feed back into the learning store. ### CONCEPT: amāna (trust, stewardship) -**Arabic:** أَمَانَة (root: أ-م-ن, safety, trust, faithfulness) +**Arabic (term and root):** أَمَانَة (root: أ-م-ن, safety, trust, faithfulness) -**Gloss:** Amāna is trust or responsibility accepted voluntarily and carrying accountability. Classical tafsir identifies it with moral accountability — the capacity to choose, and the responsibility that comes with that capacity. +**Gloss (author's):** Amāna is trust or responsibility accepted voluntarily and carrying accountability. Classical tafsir identifies it with moral accountability — the capacity to choose, and the responsibility that comes with that capacity. -**Design principle (load-bearing):** When an agent is granted access to a codebase, it accepts an amāna. The architecture must encode: (a) least-privilege defaults, (b) reversibility of every action, (c) transparency through logged rationale, (d) scope-boundedness without self-expansion of permissions. The agent is a trustee, not an owner. +**Design principle — author's analogy (load-bearing):** When an agent is granted access to a codebase, it accepts an amāna. The architecture must encode: (a) least-privilege defaults, (b) reversibility of every action, (c) transparency through logged rationale, (d) scope-boundedness without self-expansion of permissions. The agent is a trustee, not an owner. --- diff --git a/research/empirical-refutation/README.md b/research/empirical-refutation/README.md index 2c64c6ad..3f6a9532 100644 --- a/research/empirical-refutation/README.md +++ b/research/empirical-refutation/README.md @@ -2,8 +2,11 @@ *Static Impact Analysis Does Not Transfer: A Pre-Registered Refutation of Two LLM-Agent Reliability Mechanisms* -> **Corrected 2026-09-21** — see [Corrections](#corrections-2026-09-21) at the end. The PDFs in -> this directory predate the corrections; the LaTeX and HTML sources carry them. +> **Corrected 2026-09-21 and 2026-09-26** — see [Corrections](#corrections-2026-09-21) and +> [Corrections (2026-09-26)](#corrections-2026-09-26) at the end. The PDFs in this directory +> (`paper.pdf`, git blob `f94a727`; `extended_preprint.pdf`, git blob `74f74ae`) are historical, +> pre-correction editions; the LaTeX and HTML sources carry the corrections, and +> [`../HISTORICAL_EDITIONS.md`](../HISTORICAL_EDITIONS.md) records each edition's provenance. This package contains everything needed to check every number in the paper. It is organised so that a reviewer can start from the frozen protocol and work forward, in the order the work was actually @@ -48,7 +51,7 @@ append-only addenda; three were filed, all documenting the corpus-selection funn | File | What it is | |---|---| | `impact_oracle_v1_as_shipped.zip` | The version whose claims the paper refutes. 36 tests. | -| `impact_oracle_v2_src.zip` | The repaired version. 49 tests, including the stdlib-collision safety case. | +| `impact_oracle_v2_src.zip` | The repaired version. 49 tests, including the stdlib-collision safety case. The same repaired source now also lives in-tree at [`../python-prototypes/impact_oracle/`](../python-prototypes/impact_oracle/) (36 demo-package tests + 13 repair tests). | | `router_gate_src.zip` | The router and assumption gate, thresholds exactly as evaluated. 19 tests. | Each package runs with `python -m pytest` from its own root (a `conftest.py` handles the path). @@ -69,11 +72,16 @@ The 3.19% figure is what bounds achievable recall at 96.88%. **The repair.** `results/repair_results.json` → `d_defect1and2_heldout_HEADLINE`. The paper headlines `metrics_at_canonical_0.02` (F1 0.416), not the higher `metrics_at_best_threshold` (0.428), because -the latter's threshold was selected on the tuning repositories. +the latter's threshold was selected on the tuning repositories. Keep the two thresholds apart: the +per-repository counts in `per_repo_metrics_at_best_threshold` are at t = 0.10 (pooled ΔF1 over grep ++0.0565), and the package has no per-repository counts at the headline t = 0.02 (pooled ΔF1 about ++0.044), so a per-repository or sign-test statement is a statement about t = 0.10. **The held-out collapse.** `results/heldout_results.json` → `tuned_vs_heldout_comparison`. Note `cost_analysis.n_execution_verified = 0`: no held-out task admitted execution-based -verification, so correctness used a weaker model-based criterion. +verification, so correctness used a weaker model-based criterion. Every "correct" or "accepted" in +the held-out results means **judge-accepted** (`judge_accepted`), never `tests_passed`, +`human_accepted` or `deployed_without_revert`. **The cost inversion.** `results/heldout_results.json` → `cost_analysis` carries four figures along two orthogonal axes, and the paper reports all four rather than the most favourable one. Framing: @@ -94,7 +102,9 @@ than resting on the protocol's authority. Co-change is a proxy for semantic impact and errs in both directions: files co-change for reasons no static analysis can predict, and an over-warning may be a correct dependency that has not yet -co-changed. The 96.9% ceiling is measured on the graph the as-shipped oracle builds, and reachability +co-changed. It measures historically related edits, not semantic necessity or test breakage, so this +package evaluates one task — predicting co-edited files — and says nothing about the other task an +impact tool serves, selecting the tests that catch a behaviour regression. The 96.9% ceiling is measured on the graph the as-shipped oracle builds, and reachability in a dense graph is a weak property — it bounds what any static method could attain, and is not evidence that a reachable pair is causally related. @@ -132,3 +142,50 @@ WeasyPrint toolchain was available to rebuild them), and the copies of the paper Re-derive them with [`../recompute_corrections.py`](../recompute_corrections.py) (standard-library Python): extract this package and run `python research/recompute_corrections.py /repro`. + +## Corrections (2026-09-26) + +A second external deep review (2026-09-26, pinned at commit +`d2abfa69fb77531199ffc67c5c076b524af69040`) recomputed the archived results again. The counts +reproduced exactly; the corrections below sharpen what they are evidence *for*. Numbers marked +"review" were recomputed by the 2026-09-26 external review and are reproduced by +[`../recompute_corrections.py`](../recompute_corrections.py). + +- **The unit of the impact study.** 801 labelled files, 759 evaluated after the cap, nine + repositories, 20,144 mirrored labelled pairs (review). Original oracle pooled P / R / F1 + **0.3982 / 0.0220 / 0.0416**; grep **0.3535 / 0.5732 / 0.4373** (review). Repository-cluster + bootstrap, 20,000 draws, seed 1234: oracle F1 **[0.0010, 0.0927]**, grep **[0.3807, 0.5394]**, + grep minus oracle **[0.3422, 0.5174]** (review). Repository-level view: macro F1 0.0220 against + 0.4947, grep ahead in 9 of 9 repositories (recompute script §5). The negative result is well + supported within this corpus. What it measures is co-edited-file prediction on a proxy label: + co-change is historically related editing, not semantic necessity or test breakage; results for + (a) co-edited-file prediction and (b) regression-test selection must be reported separately, and + (b) was not measured. Mirrored pairs and shared files are not independent examples, which is why + uncertainty is reported by repository and macro beside pooled. The Node regex graph in + `src/atlas.js` is not the evaluated Python AST oracle and inherits none of these numbers. +- **The repair, one threshold at a time.** At the headline threshold 0.02 the repaired oracle's F1 + is 0.416 against grep's 0.371 (ΔF1 about +0.0442). Per-repository counts exist only at threshold + 0.10, where the pooled ΔF1 is +0.0565, pytest supplies 71.3% of the held-out pairs, and 3 of 3 + held-out repositories favour the repair with a one-sided sign-test p = 0.125 (review). The eight + numeric parameters were frozen on six tuning repositories, but the decision to add the sibling and + forward relations followed diagnosis across all nine: an architecture-selection channel into the + nominal test set. It does not erase the measured improvement; it limits the unseen-repository + claim. The next study freezes parser and relation design before a new repository set or time split + is acquired, adds runtime coupling, configuration changes, dynamic imports and non-Python + languages, predeclares relation budgets, and reports files reviewed per true affected file and the + missed-regression rate beside F1. +- **The router's success metric is judge acceptance.** On the 64 non-halted held-out tasks the + routed pipeline spent **$6.3582** against always-premium's **$5.2893**: **20.21% more**, not saved. + Judge-accepted outputs were **6/64** against **3/64**, so cost per judge-accepted output is + **$1.060** against **$1.763**, a ratio that is unstable at those counts. The judge model is also + the mid-tier executor, and re-labelling 30 tasks with the same model and a reworded prompt gives + halt κ **0.5161** and tier κ **0.8919** (review): self-consistency, not independent human + agreement. Coding success should be anchored by `tests_passed` (executable tests) and blind + `human_accepted` adjudication of disagreements, with `deployed_without_revert` where available; + none of these was measured here. Gate precision/recall, router solve rate and total pipeline cost + are separate endpoints, and clarification turns and rejected-but-valid tasks should be counted. + +**Downloads.** `paper.pdf` and `extended_preprint.pdf` are pre-correction editions (corrected +sources: `paper/main.tex`, `extended_preprint.html`); `replication_package.tar.gz` (git blob +`50bd453`) is left exactly as published, and the corrected summary of what it shows is this README's +two Corrections sections. diff --git a/research/empirical-refutation/extended_preprint.html b/research/empirical-refutation/extended_preprint.html index 889e11f1..643f9a75 100644 --- a/research/empirical-refutation/extended_preprint.html +++ b/research/empirical-refutation/extended_preprint.html @@ -82,14 +82,13 @@

      A Formal Theory of the Cognitive Substrate for Coding Agents

      Extended edition, with a pre-registered empirical refutation. Unifies the substrate faculties, the end-to-end reliability framework, and the forgekit implementation — with a two-layer duality theorem, a unified algorithm set, a Qur’anic epistemology carried in full, and a measurement that overturned two of this work’s own headline claims.

      - +

      Status of this edition

      This is the extended companion to a venue submission reporting a pre-registered empirical evaluation of the two prototypes described here. That evaluation refuted both of their headline claims: the impact oracle's perfect recall collapsed from 1.00 to 0.022 on real repositories, and the router/gate -pair's perfect separation fell to F1 = 0.37 with its cost saving inverting from +62.1% to −20.2% once the pipeline's escalation retries are counted. Section 10 reports the refutation, the diagnosis, and a repair that recovers a -narrow win over the baseline, and states what the failure costs the formalism — specifically, that +pair's perfect separation fell to F1 = 0.37 with its cost saving inverting from +62.1% to −20.2% once the pipeline's escalation retries are counted. Section 10 reports the refutation, the diagnosis, and a repair whose point estimate is above the baseline without being shown to beat it [corrected 2026-09-26], and states what the failure costs the formalism — specifically, that Theorem T5's completeness guarantee transfers nothing to practice until the underlying relation is shown adequate. Theory sections are otherwise unchanged; where they make empirical claims, those claims are now the corrected ones.

      @@ -100,9 +99,14 @@

      Corrections (2026-09-21)

      An external review found that Theorem D was circular as stated, that Eq. (5) assumed an independence the design contradicts, that several definitions and proofs were wrong, that one sentence still claimed the prototype’s perfect recall, and that some statistical inferences in §10 were stronger than the data support. The theory sections are therefore no longer unchanged from the synthesis edition. The corrections are made in place, marked [corrected 2026-09-21], and listed with the original wording in Corrections. The PDF edition predates them.

      +
      +

      Corrections (2026-09-26)

      +

      A second external review (2026-09-26) found that Theorem D's range statement combined two maxima that need not be attainable together, that an equality condition was misstated, that a caught miss was being read as a completed task, that the frozen-map premise was broader than the guarantees it actually removes, and that the prior art and the shared authorship of the “three bodies of work” needed the same care already given to priority. The corrections are made in place, marked [corrected 2026-09-26], and listed with the original wording in Corrections (2026-09-26). The PDF edition predates both sets of corrections.

      +
      +

      Abstract

      -

      A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

      +

      A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. Prompting does change behaviour within a context; what it cannot supply on its own is durable state across invocations and a check the model does not grade itself, and this architecture is one tested way of supplying both, not the only possible one [corrected 2026-09-26]. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

      @@ -122,6 +126,7 @@

      Abstract

    1. Honest limits — what no architecture can guarantee
    2. Conclusion
    3. Corrections (2026-09-21)
    4. +
    5. Corrections (2026-09-26)
    6. Appendix A: graded reference set  ·  Appendix B: crosswalk table  ·  References
      @@ -141,6 +146,7 @@

      1 The convergence — three roads to one architectureThe claim of this paper

      These are not three similar ideas. They are one architecture described in three vocabularies. The impact-awareness faculty is the change-closure fixpoint. The assumption gate is the amnesia equation. The substrate's external structure is a two-layer duality — and that duality, which the reliability framework states as a design law, is the result the whole thing turns on. What each road saw partially, the union sees whole.

      Two of these three “identities” are weaker than this callout says: the impact oracle approximates Δ* rather than computing it (§3.2), and the duality is a bound over a region of parameters, not a theorem for every p, c < 1 (§4). See the Corrections. [corrected 2026-09-21]

      +

      The five faculties are one useful decomposition, not a proof that these five are necessary or that an external stateful architecture is the only way to supply them. The broad architecture has clear prior art: CoALA (Sumers et al., 2023) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023) improves agents through linguistic feedback and an episodic memory buffer with no weight updates. The defensible claim is narrower: a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes. [corrected 2026-09-26]

      The synthesis also inherits a governing discipline, stated plainly by the practitioner who commissioned this work: AI output is a mathematically calculated probability; it must never be trusted blindly; for the same prompt it can give a different answer, so use only the capability it is genuinely best at, and earn trust with an external check. We will see that this sentence is not a slogan but the informal statement of the central theorem — the quantity (1−p)>0 that forces a deterministic layer to exist.

      @@ -155,7 +161,7 @@

      2 The object of study — the frozen map and its five la
      • P1 — statelessness. fθ has no memory across calls; each invocation sees only the current x. Nothing the agent learned yesterday is present today unless something outside the model re-supplies it.
      • -
      • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; any learning must be external.
      • +
      • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; learning that has to outlast the current context must be held outside the model. [corrected 2026-09-26]
      • P3 — bounded, undifferentiated context. x is finite and flat: a long story and a long program are the same kind of object to it, with no privileged channel for goals versus detail. This is the root of goal-drift and of context saturation.
      @@ -172,7 +178,7 @@

      2 The object of study — the frozen map and its five la

      The “Forced by” column now matches the whitepaper's derivation, which argues each row separately; the earlier version of this table disagreed with it in all five rows. [corrected 2026-09-21]

      -

      The critical word is external. Because θ is frozen (P2) and context is bounded (P3), none of these can be fixed by prompting harder or by fine-tuning alone. The architecture must live around the model, hold state outside it, and enforce behaviour the model cannot be relied upon to produce on its own. The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

      +

      The critical word is external. What P1–P3 remove is a set of guarantees, not every behaviour: there is no built-in durable state across independent invocations, the context is bounded, parameters are not updated automatically from outcomes, and self-verification without external evidence is unreliable. Examples, retrieved facts and feedback placed in the context do change behaviour with no weight update — GPT-3's few-shot evaluation measures exactly that adaptation through text (Brown et al., 2020) — so prompting is not powerless. What it cannot supply on its own is persistence beyond the window and a check the model does not grade itself. The architecture supplies those: it lives around the model, holds state outside it, and enforces behaviour the model cannot be relied upon to produce on its own. It is one tested way of supplying persistence and verification, not the only logically possible architecture. [corrected 2026-09-26] The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

      3 Definitions

      @@ -261,12 +267,12 @@

      4 The central result — the two-layer duality theorem

      With cj = P(check j fires | M), and no independence assumption, max(0, 1−Σjcj) ≤ 1−q ≤ 1−maxjcj (Fréchet bounds). The product (1−p)·∏j(1−cj), which earlier versions gave as Eq. (5), is the special case in which the checks fire independently given the miss. When the checks are nested, for example the same classifier run at several points on the same diff, r = (1−p)(1−cmax).

      Then:

        -
      1. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). It equals 1−(1−ε)n only if tasks fail independently, and tasks done by one model on one repository need not.
      2. +
      3. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). If tasks fail independently with per-task residuals ri ≤ ε, then P(≥1 miss) = 1−∏i(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε: independence alone does not give equality. Tasks done by one model on one repository need not be independent at all. [corrected 2026-09-26]
      4. Instruction layer alone (q = 0): r = 1−p, so reaching ε needs p ≥ 1−ε from instructions. Raising p does bend the curve: for 30 independent tasks, P(≥1 miss) is 0.958 at p = 0.9 and 0.260 at p = 0.99.
      5. Deterministic layer alone (the bare model's p0): reaching ε needs q ≥ 1 − ε/(1−p0).
      6. Composition. Adding a check with P(it fires | M, no earlier check fired) > 0 strictly lowers r. Adding a copy of a check that is already present lowers nothing.
      -

      The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they can reach (1−pmax)(1−qmax). So a target ε with (1−pmax)(1−qmax) ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. Whether a real target falls in that range is an empirical question about p0, pmax and qmax, which this paper does not measure.

      +

      The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they reach r* = min(p,q)∈F (1−p)(1−q) over the joint feasible set F = {(p(π), q(π)) : π an admissible policy}. Instructions change which misses remain, and q is a catch rate conditional on that changed miss population, so pmax and qmax need not be attainable under one policy: (1−pmax)(1−qmax) is a lower bound on r*, reached only if the two maxima are jointly attainable on the same task distribution. For example, policy A with (p, q) = (0.5, 0.9) leaves 0.05 and policy B with (0.9, 0.1) leaves 0.09; the separate maxima, 0.9 and 0.9, suggest 0.01, which neither policy attains (research/recompute_corrections.py asserts this in §3b). So a target ε with r* ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. [corrected 2026-09-26] Whether a real target falls in that range is an empirical question about p0, pmax, qmax and the shape of F, which this paper does not measure.

      □

      @@ -277,7 +283,7 @@

      4 The central result — the two-layer duality theorem
      The two-layer duality architecture -
      Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Neither layer alone suffices — the formal content of the discipline never trust the output; earn trust with a check.
      +
      Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Where each factor is bounded away from zero, neither layer alone reaches a small residual [corrected 2026-09-26] — the formal content of the discipline never trust the output; earn trust with a check.
      @@ -305,6 +311,9 @@

      The honest cost side

      +

      5.4 A caught miss is not a completed task [corrected 2026-09-26]

      +

      Theorem D counts silent misses. A miss that a check catches is not thereby a completed, correct task: the gate can block the same turn repeatedly, the agent can abandon the task, and the repair can fail. So a lower silent-miss probability is not automatically a higher completed-correct-task rate, and the value of the two layers has to be measured on outcomes rather than read off r. The quantities that decide it are the true catch rate, the false-block rate, repaired success conditional on a catch, abandonment, added latency and recovery cost. This paper measures none of them.

      +

      6 The unified algorithm set — the TASK loop

      The faculties of Def. 4 are realized by seven algorithms. They are the reliability framework's A1–A7, recast here as the operations of the substrate: each is a faculty made mechanical, each binds to one lifecycle point, and together they form a single loop whose progress is guaranteed by an explicit worklist and whose floor is guaranteed by a deterministic gate.

      @@ -614,7 +623,7 @@

      What this does to Theorem T5 — the correction that matters

      Concept / verseRetrieved glossFacultyDesign principle (abbrev.)Type
      Concept / verseTranslation (verses) / author’s gloss (concepts)FacultyAuthor’s design analogy (abbrev.)Type
      17:36 — lā taqfu (do not pursue without knowledge)Do not follow blindly what you do not know to be true: ears, eyes, and heart, you will be questioned about all these.IMPACT-AWARENESSBefore any code mutation (file write, delete, refactor), the agent must run a pre-action verification gate that checks: (1) what entities in the codebase wil…load-bearing
      49:6 — tabayyun (verify reports before acting)Believers, if a troublemaker brings you news, check it first, in case you wrong others unwittingly and later regret what you have done,SELF-CORRECTIONThe agent architecture must include a verification gate between receiving information (from context, tool output, or its own prior reasoning) and acting on itload-bearing
      grep baseline, held-out0.2690.6010.371
      -

      The repaired oracle's point estimate is above the baseline for the first time, reaching 66.8% of the static ceiling. Earlier versions said it beats the baseline; that is not established. [corrected 2026-09-21] F1 is higher by 0.044, and all three held-out repositories agree in sign, but three out of three gives a one-sided sign-test p of 0.125, pytest supplies 71% of the held-out pairs, and the file-level F1 intervals overlap. The choice of which relations to add also came from failure analysis pooled over all nine repositories, so the split was clean for the numeric parameters but not for that structural choice. Two details are worth more than the headline. First, the +

      The repaired oracle's point estimate is above the baseline for the first time, reaching 66.8% of the static ceiling. Earlier versions said it beats the baseline; that is not established. [corrected 2026-09-21] F1 is higher by 0.044 at the canonical threshold 0.02; at threshold 0.10, the only threshold with per-repository counts in the package, the pooled gain is 0.057 and all three held-out repositories agree in sign [corrected 2026-09-26], but three out of three gives a one-sided sign-test p of 0.125, pytest supplies 71% of the held-out pairs, and the file-level F1 intervals overlap. The choice of which relations to add also came from failure analysis pooled over all nine repositories, so the split was clean for the numeric parameters but not for that structural choice. Two details are worth more than the headline. First, the obvious repair of the construction defect is unsafe — it fabricates dependency edges through standard-library name collisions — so we applied a more conservative fix with a smaller gain (11.0× rather than 14.5×); a tool that invents edges to raise recall is worse than one that misses @@ -648,7 +657,7 @@

      10.2 Prototype II — the router and gate, refuted

      The gate missed roughly seven in ten under-specified requests. Routing retained partial signal — within-one-tier accuracy of 0.91 is well above chance, so the complexity rubric measures something -— but exact-tier accuracy fell to 0.53, and the cost saving did not merely shrink but inverted: routing does save 59.5% in raw dollars on first attempts alone, but almost none of that cheaper output is correct (3.6% once gated), and counting what the pipeline actually spent escalating up the tier ladder, it costs 20.2% more than always using the premium tier. Labelling noise is real and reported rather than hidden: agreement between two labelling passes on the +— but exact-tier accuracy fell to 0.53, and the cost saving did not merely shrink but inverted: routing does save 59.5% in raw dollars on first attempts alone, but almost none of that cheaper output is judged correct (3.6% once gated on the judge's acceptance; the judge is a model, not an executed test) [corrected 2026-09-26], and counting what the pipeline actually spent escalating up the tier ladder, it costs 20.2% more than always using the premium tier. Labelling noise is real and reported rather than hidden: agreement between two labelling passes on the should-ask label was κ = 0.52, moderate, which bounds how well any gate could score here. Both passes were the same model with differently worded prompts, and that model also judged correctness, so κ measures robustness to prompt wording, not label validity. The cost inversion is also largely mechanical: 58 of 64 tasks failed at every tier and always-premium was judged correct on only 3, so escalation paid for every tier. Per output the judge accepted, the pipeline cost $1.06 against always-premium's $1.76, from 6 and 3 accepted outputs. [corrected 2026-09-21]

      @@ -689,7 +698,7 @@

      12 Honest limits — what no architecture can guarantee

      13 Conclusion

      -

      A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the faculties it structurally lacks. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

      +

      A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the persistence and external checks it does not provide on its own [corrected 2026-09-26]. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

      The limit this edition discovered the hard way

      The honest limits listed below were all stated before any real-repository measurement existed. One more @@ -730,8 +739,21 @@

      Corrections (2026-09-21)

    +

    Corrections (2026-09-26)

    +

    A second external deep review of the forgekit repository (2026-09-26, pinned at commit d2abfa69fb77531199ffc67c5c076b524af69040) re-checked the corrected Theorem D and this paper's framing. Each change is made in place above, marked [corrected 2026-09-26], and listed here with the original wording, so nothing is silently rewritten. The counterexample in item 1, the equality condition in item 2 and the 400-fold correction in item 4 are asserted by python3 research/recompute_corrections.py --theorem-checks (§3b), which needs no data. The PDF edition predates these corrections as well.

    +
      +
    1. Separately maximal rates need not be jointly attainable (§4). The range statement read: “together they can reach (1−pmax)(1−qmax)”. That requires both maxima to be attainable under one policy on the same task distribution; instructions change which misses remain, and q is conditional on that population. The attainable residual is min(p,q)∈F (1−p)(1−q) over F = {(p(π), q(π)) : admissible π}, and the product of the separate maxima is only a lower bound on it unless compatibility is established. Counterexample: policy A (0.5, 0.9) leaves 0.05, policy B (0.9, 0.1) leaves 0.09, and the separate maxima suggest 0.01, which neither attains.
    2. +
    3. Equality in the independent-task bound (§4, Claim 1). The text said P(≥1 miss) “equals 1−(1−ε)n only if tasks fail independently”. From per-task residuals ri ≤ ε, independence gives 1−∏(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε. With residuals 0.01, 0.005 and 0.001, for example, the probability is 0.0159 against the bound's 0.0297.
    4. +
    5. A caught miss is not a completed task (new §5.4). A lower silent-miss probability does not by itself mean more completed, correct tasks: a caught mistake can end in an abort, repeated blocking or a failed repair. The measures that decide the question — true catch rate, false-block rate, repaired success conditional on a catch, abandonment, latency and recovery cost — are listed and are not measured here.
    6. +
    7. Three lifecycle copies of one classifier are one detector (§5.3). The 2026-09-21 correction stands: multiplying identical checks at the Stop hook, pre-commit and CI as if independent understated the residual 400-fold. Copies at several points can widen the opportunities to run a check, such as edits made after a turn ended, but they are not independent semantic detectors.
    8. +
    9. The frozen-map premise, stated as the guarantees it removes (abstract, §2). The paper said “none of these can be fixed by prompting harder or by fine-tuning alone” and “any learning must be external”. Frozen parameters exclude weight updates during use; they do not exclude changed behaviour from examples, retrieved facts, feedback or additional computation in the current context, which GPT-3's few-shot evaluation measured with no gradient updates (Brown et al., 2020, arXiv:2005.14165). The missing guarantees are narrower: no built-in durable state across independent invocations; a bounded context; no automatic parameter update; and unreliable self-verification without external evidence. The substrate is one tested way of supplying persistence and verification, not the only logically possible architecture.
    10. +
    11. Prior art and what is (and is not) claimed (byline, abstract, §1, §13). External memory, feedback-driven improvement and structured agent control have clear prior art: CoALA (Sumers et al., 2023, arXiv:2309.02427) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023, arXiv:2303.11366) improves agents with linguistic feedback and an episodic memory buffer and no weight updates. Both were already among this paper's references. The defensible claim is a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes; novelty is claimed only for a specific protocol, invariant, evaluation result or integration that survives an explicit prior-art comparison. The five faculties are a decomposition, not a proof of necessity, and the conclusion no longer says the architecture “supplies the faculties it structurally lacks”. The byline still called the three bodies of work “independently-developed” after the 2026-09-21 corrections had shown that they share an author; it now says so.
    12. +
    13. Reference grades are bibliographic (Appendix A). confirmed and traceable record that a source exists and is correctly attributed. They are not a judgement that the source supports the theory it is cited for. graded_reference_set.md now keeps bibliographic verification, claim support, study design, independent replication and transfer scope as separate grades.
    14. +
    15. Two statements in §10 and the status box. (a) The status box said the repair “recovers a narrow win over the baseline”; §10 had already withdrawn that on 2026-09-21, and the box now matches it. (b) §10.1 put the canonical-threshold gain, 0.044 at t = 0.02, next to the sign agreement of the three held-out repositories, which is only computable at t = 0.10 (pooled gain 0.057), the one threshold with per-repository counts in the package. The two thresholds are now reported separately. (c) §10.2 said almost none of the cheaper output “is correct”; correctness there is a model judge's acceptance (the judge is also the mid-tier executor), not executed tests, and the text now says judge-accepted.
    16. +
    +

    Appendix A — Graded reference set (new sources)

    -

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21]

    +

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21] These grades are bibliographic: they confirm that a record exists and is correctly attributed. They do not grade whether a source supports the claim it is cited for, its study design, independent replication or transfer scope; research/formal-synthesis/graded_reference_set.md now keeps those as separate fields. [corrected 2026-09-26]

    diff --git a/research/formal-synthesis/README.md b/research/formal-synthesis/README.md index c44eecbc..af351904 100644 --- a/research/formal-synthesis/README.md +++ b/research/formal-synthesis/README.md @@ -8,6 +8,15 @@ > statement. **`substrate_synthesis.pdf` predates these corrections** (it was built with > WeasyPrint, which was not available to rebuild it); read the HTML. The numbers are > recomputed by [`../recompute_corrections.py`](../recompute_corrections.py). +> +> **Corrected again 2026-09-26.** A second review found that the corrected range statement still +> combined two maxima that need not be attainable together, `(1 − p_max)(1 − q_max)`; that the equality condition +> for `1 − (1 − ε)ⁿ` was misstated; that a caught miss was being read as a completed task; that the +> frozen-map premise was broader than the guarantees it removes; and that prior art and the shared +> authorship of the "three bodies of work" needed stating. The HTML's **Corrections (2026-09-26)** +> section lists each with its original wording, and +> `python3 ../recompute_corrections.py --theorem-checks` asserts the numeric counterexample, the +> equality condition and the 400× correction with no data. This directory contains a formal, mathematical unification of three separately developed bodies of work that all describe the **same architecture** for making a @@ -44,6 +53,20 @@ dependence between tasks. What the corrections changed, briefly: +- **The two maxima need not be jointly attainable (2026-09-26).** The attainable residual is + `r* = min over (p, q) ∈ F of (1 − p)(1 − q)`, with `F = {(p(π), q(π)) : admissible policies π}`. + Instructions change which misses remain, and `q` is a catch rate on that changed population, so + `(1 − p_max)(1 − q_max)` is a *lower bound* on `r*` unless compatibility is established. Policy A + with `(p, q) = (0.5, 0.9)` leaves 0.05 and policy B with `(0.9, 0.1)` leaves 0.09; the separate + maxima suggest 0.01, which neither attains. +- **Equality needs equal residuals (2026-09-26).** With per-task residuals `rᵢ ≤ ε` and independent + tasks, `P(≥1 miss) = 1 − ∏(1 − rᵢ) ≤ 1 − (1 − ε)ⁿ`, with equality only when every `rᵢ = ε`; + independence alone does not give equality. The union bound `nε` needs neither. +- **Catching is not completing (2026-09-26).** A lower silent-miss probability is not automatically a + higher completed-correct-task rate: a caught mistake can end in an abort, repeated blocking or a + failed repair. Measure the true catch rate, false-block rate, repaired success conditional on a + catch, abandonment, latency and recovery cost. + - **It is a bound, not an impossibility proof.** The old criterion, `P(≥1 miss) → 1`, also condemns the composed system (0.993 over 1,000 tasks at a residual of 0.005), and raising `p` bends the curve too (30-task `P(≥1 miss)` is 0.958 at `p = 0.9` and 0.260 at @@ -52,6 +75,9 @@ What the corrections changed, briefly: `(1 − p)·∏(1 − cⱼ)`, assumed the checks fire independently given a miss. The same classifier at the Stop hook, pre-commit and CI fires together, so the residual is `(1 − p)(1 − c_max)`: 0.015, not the product's 3.75 × 10⁻⁵, in the paper's own example. + Three lifecycle copies of one classifier widen the opportunities to run it; they are not three + independent semantic detectors. The 400× figure is recomputed (§2) and asserted (§3b) by the + recomputation script. - **`cⱼ` belongs to the agent as well as the gate.** The gate detects its proxy exactly, not the miss; an agent that touches `STATE.md` passes it. At a STATE-touch rate of 0.9 the residual is 0.27, not 0.015. @@ -60,6 +86,15 @@ What the corrections changed, briefly: - **Priority is conceded.** The law is standard layer-of-protection algebra, and two concurrent preprints derived a more general Bayesian form first (see the refutation paper's related work). The paper no longer says it "proves" the result. +- **The same care for the rest of the framing (2026-09-26).** A frozen model lacks specific + guarantees — durable state across independent invocations, context beyond the window, automatic + parameter update, reliable self-verification without external evidence — not the ability to adapt + inside a context (Brown et al., 2020, [arXiv:2005.14165](https://arxiv.org/abs/2005.14165)). + CoALA ([arXiv:2309.02427](https://arxiv.org/abs/2309.02427)) and Reflexion + ([arXiv:2303.11366](https://arxiv.org/abs/2303.11366)) are prior art for the broad architecture; + the defensible claim is *a portable implementation of evidence-weighted coding-agent memory and + checks, with empirical evaluation of trust failure modes*, and the five faculties are a + decomposition, not a proof of necessity. It is the formal content of the practitioner's rule: _never trust the output of a probability engine; earn trust with an external check._ @@ -79,10 +114,10 @@ set, and said reverse reachability run to fixpoint implied perfect recall. | File | What it is | | ----------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `substrate_synthesis.pdf` | The formal synthesis paper (42 pp): definitions, Theorem D + proof, the unified A1–A7 TASK loop, invariants I1–I4, theorems T1–T6 with proofs, the 16-row crosswalk, the full 14-mapping Qur'anic epistemology, both prototypes. **Predates the 2026-09-21 corrections.** | -| `substrate_synthesis.html` | Same paper, self-contained HTML, **with the 2026-09-21 corrections** and a Corrections section. | +| `substrate_synthesis.pdf` | The formal synthesis paper (42 pp): definitions, Theorem D + proof, the unified A1–A7 TASK loop, invariants I1–I4, theorems T1–T6 with proofs, the 16-row crosswalk, the full 14-mapping Qur'anic epistemology, both prototypes. **Historical, pre-correction edition** (git blob `2e17362`; predates the 2026-09-21 and 2026-09-26 corrections — see [`../HISTORICAL_EDITIONS.md`](../HISTORICAL_EDITIONS.md)). | +| `substrate_synthesis.html` | Same paper, self-contained HTML, **with the 2026-09-21 and 2026-09-26 corrections** and a Corrections section for each. The corrected source. | | `crosswalk.json` / `crosswalk.md` | The three-way term-by-term correspondence (substrate ↔ framework ↔ forgekit), with the P1/P2/P3 → Π₁/Π₂/Π₃ notation reconciliation. | -| `graded_reference_set.json` / `.md` | The 15 new sources independently verified and graded (9 confirmed, 6 traceable, 0 unverifiable), including the disambiguation of the two future-dated arXiv IDs. | +| `graded_reference_set.json` / `.md` | The 15 new sources independently verified and graded (9 confirmed, 6 traceable, 0 unverifiable), including the disambiguation of the two future-dated arXiv IDs. These are bibliographic grades (the source exists and is correctly attributed); claim support, study design, replication and transfer scope are separate and not assessed. | | `merged_references.json` | Full 47-entry bibliography (32 original + 15 new, deduped). | | `figures/schematic_duality.png` | The two-layer duality architecture. | | `figures/schematic_taskloop.png` | The unified 7-stage TASK loop (each stage bound to faculty · algorithm · Qur'anic anchor). | @@ -91,15 +126,18 @@ The **two runnable prototypes** referenced throughout the paper already live in repo and are not duplicated here: - `../python-prototypes/impact_oracle/` — Prototype I, the impact oracle (approximates - A1 / Δ\*). Runnable, 36 tests. Recall 1.00 on five mutations of its own demo package; + A1 / Δ\*). Runnable; the in-tree package is the repaired v2 (49 tests: 36 demo-package + 13 + repair; corrected 2026-09-26 from "36 tests"). Recall 1.00 on five mutations of its own demo package; **refuted on real repositories: recall 0.022** on 759 files in nine repositories, where grep scored F1 0.437 against the oracle's 0.042 (see [`../empirical-refutation/`](../empirical-refutation/)). - `../python-prototypes/router_gate/` — Prototype II, complexity-router + - assumption-gate (A7 + A6 / M1 + M2). Runnable, 19 tests. 62.1% cost saved on the 30 + assumption-gate (A7 + A6 / M1 + M2). Runnable, 23 tests in-tree (the version archived as + evaluated has 19; corrected 2026-09-26). 62.1% cost saved on the 30 tasks its thresholds were tuned on; **refuted on 80 held-out tasks: total spend was - 20.2% higher** than always-premium. Per output a judge accepted, it cost $1.06 against - always-premium's $1.76, but only 6 and 3 of 64 outputs were accepted. + 20.2% higher** than always-premium. Per judge-accepted output (a model judge that is also the + mid-tier executor; no output was test-verified), it cost $1.06 against always-premium's $1.76, + but only 6 and 3 of 64 outputs were accepted. ## Honesty commitments (carried from the source work) diff --git a/research/formal-synthesis/graded_reference_set.json b/research/formal-synthesis/graded_reference_set.json index eea89e1a..76f0ccc3 100644 --- a/research/formal-synthesis/graded_reference_set.json +++ b/research/formal-synthesis/graded_reference_set.json @@ -6,6 +6,17 @@ "traceable": "the work clearly exists and is correctly attributed, but rests on established secondary/historical knowledge rather than a single retrievable record", "unverifiable": "could not confirm existence or attribution" }, + "grade_scope": { + "added": "2026-09-26", + "grade_is": "bibliographic_verification", + "meaning": "confirmed / traceable / unverifiable establish that a source exists and is correctly attributed; they are not independent validation of the theory it is cited for.", + "not_assessed": [ + "claim_support", + "study_design", + "independent_replication", + "transfer_scope" + ] + }, "tally": { "confirmed": 9, "traceable": 6, diff --git a/research/formal-synthesis/graded_reference_set.md b/research/formal-synthesis/graded_reference_set.md index cb58d82e..10faf3e4 100644 --- a/research/formal-synthesis/graded_reference_set.md +++ b/research/formal-synthesis/graded_reference_set.md @@ -6,6 +6,22 @@ Independent verification: modern arXiv sources by direct metadata API fetch (tit **Tally: 9 confirmed · 6 traceable · 0 unverifiable** (of 15 new sources). +**What these grades are (added 2026-09-26).** `confirmed` / `traceable` / `unverifiable` are +**bibliographic verification** grades: they establish that a source exists and is correctly +attributed. They are not independent validation of the theory the source is cited for. Four other +dimensions are kept separate and are **not assessed** for these 15 sources: + +| Dimension | Question it answers | Status for this set | +|---|---|---| +| Bibliographic verification | Does the source exist and is it correctly attributed? | graded below | +| Claim support | Does the source support the specific claim it is cited for? | not assessed | +| Study design | What kind of evidence is it (experiment, survey, position paper, book, doctrine)? | not assessed | +| Independent replication | Has anyone other than the authors reproduced the result? | not assessed | +| Transfer scope | Which settings can the result be carried to? | not assessed | + +Most of these sources are cited as architectural analogues or framing (CoALA, ReAct, OODA, PDCA), so a +`confirmed` grade here says the citation is real, not that the synthesis is right. + | Source | ID | Grade | Note | |---|---|---|---| | **Cognitive Architectures for Language Agents** (Theodore R. Sumers, Shunyu Yao, Karthik Narasimhan et al., 2023) | `2309.02427` | confirmed | Retrieved via arXiv metadata API; title/authors match claim exactly. Unifies memory, planning/reasoning, action, and learning modules into a single CoALA framework for language agents, giving the c… | diff --git a/research/formal-synthesis/substrate_synthesis.html b/research/formal-synthesis/substrate_synthesis.html index 4ce4d124..bdce98f9 100644 --- a/research/formal-synthesis/substrate_synthesis.html +++ b/research/formal-synthesis/substrate_synthesis.html @@ -82,11 +82,11 @@

    A Formal Theory of the Cognitive Substrate for Coding Agents

    Unifying the substrate faculties, the end-to-end reliability framework, and the forgekit implementation — with a two-layer duality theorem, a unified algorithm set, and a Qur'anic epistemology.

    - +

    Abstract

    -

    A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

    +

    A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. Prompting does change behaviour within a context; what it cannot supply on its own is durable state across invocations and a check the model does not grade itself, and this architecture is one tested way of supplying both, not the only possible one [corrected 2026-09-26]. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

    @@ -94,6 +94,11 @@

    Corrections (2026-09-21)

    An external review found that Theorem D was circular as stated, that Eq. (5) assumed an independence the design contradicts, that several definitions and proofs were wrong, and that the prototype results in §10 had already been refuted. The corrections are made in place, marked [corrected 2026-09-21], and listed with the original wording in Corrections. The PDF edition predates them.

    +
    +

    Corrections (2026-09-26)

    +

    A second external review (2026-09-26) found that Theorem D's range statement combined two maxima that need not be attainable together, that an equality condition was misstated, that a caught miss was being read as a completed task, that the frozen-map premise was broader than the guarantees it actually removes, and that the prior art and the shared authorship of the “three bodies of work” needed the same care already given to priority. The corrections are made in place, marked [corrected 2026-09-26], and listed with the original wording in Corrections (2026-09-26). The PDF edition predates both sets of corrections.

    +
    +
    Contents
      @@ -112,6 +117,7 @@

      Corrections (2026-09-21)

    1. Conclusion
    2. Four arrivals by one author — the convergence audited
    3. Corrections (2026-09-21)
    4. +
    5. Corrections (2026-09-26)
    Appendix A: graded reference set  ·  Appendix B: crosswalk table  ·  References
    @@ -131,6 +137,7 @@

    1 The convergence — three roads to one architectureThe claim of this paper

    These are not three similar ideas. They are one architecture described in three vocabularies. The impact-awareness faculty is the change-closure fixpoint. The assumption gate is the amnesia equation. The substrate's external structure is a two-layer duality — and that duality, which the reliability framework states as a design law, is the result the whole thing turns on. What each road saw partially, the union sees whole.

    Two of these three “identities” are weaker than this callout says: the impact oracle approximates Δ* rather than computing it (§3.2), and the duality is a bound over a region of parameters, not a theorem for every p, c < 1 (§4). See the Corrections. [corrected 2026-09-21]

    +

    The five faculties are one useful decomposition, not a proof that these five are necessary or that an external stateful architecture is the only way to supply them. The broad architecture has clear prior art: CoALA (Sumers et al., 2023) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023) improves agents through linguistic feedback and an episodic memory buffer with no weight updates. The defensible claim is narrower: a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes. [corrected 2026-09-26]

    The synthesis also inherits a governing discipline, stated plainly by the practitioner who commissioned this work: AI output is a mathematically calculated probability; it must never be trusted blindly; for the same prompt it can give a different answer, so use only the capability it is genuinely best at, and earn trust with an external check. We will see that this sentence is not a slogan but the informal statement of the central theorem — the quantity (1−p)>0 that forces a deterministic layer to exist.

    @@ -145,7 +152,7 @@

    2 The object of study — the frozen map and its five la
    • P1 — statelessness. fθ has no memory across calls; each invocation sees only the current x. Nothing the agent learned yesterday is present today unless something outside the model re-supplies it.
    • -
    • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; any learning must be external.
    • +
    • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; learning that has to outlast the current context must be held outside the model. [corrected 2026-09-26]
    • P3 — bounded, undifferentiated context. x is finite and flat: a long story and a long program are the same kind of object to it, with no privileged channel for goals versus detail. This is the root of goal-drift and of context saturation.
    @@ -162,7 +169,7 @@

    2 The object of study — the frozen map and its five la

    The “Forced by” column now matches the whitepaper's derivation, which argues each row separately; the earlier version of this table disagreed with it in all five rows. [corrected 2026-09-21]

    -

    The critical word is external. Because θ is frozen (P2) and context is bounded (P3), none of these can be fixed by prompting harder or by fine-tuning alone. The architecture must live around the model, hold state outside it, and enforce behaviour the model cannot be relied upon to produce on its own. The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

    +

    The critical word is external. What P1–P3 remove is a set of guarantees, not every behaviour: there is no built-in durable state across independent invocations, the context is bounded, parameters are not updated automatically from outcomes, and self-verification without external evidence is unreliable. Examples, retrieved facts and feedback placed in the context do change behaviour with no weight update — GPT-3's few-shot evaluation measures exactly that adaptation through text (Brown et al., 2020) — so prompting is not powerless. What it cannot supply on its own is persistence beyond the window and a check the model does not grade itself. The architecture supplies those: it lives around the model, holds state outside it, and enforces behaviour the model cannot be relied upon to produce on its own. It is one tested way of supplying persistence and verification, not the only logically possible architecture. [corrected 2026-09-26] The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

    3 Definitions

    @@ -251,12 +258,12 @@

    4 The central result — the two-layer duality theorem

    With cj = P(check j fires | M), and no independence assumption, max(0, 1−Σjcj) ≤ 1−q ≤ 1−maxjcj (Fréchet bounds). The product (1−p)·∏j(1−cj), which earlier versions gave as Eq. (5), is the special case in which the checks fire independently given the miss. When the checks are nested, for example the same classifier run at several points on the same diff, r = (1−p)(1−cmax).

    Then:

      -
    1. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). It equals 1−(1−ε)n only if tasks fail independently, and tasks done by one model on one repository need not.
    2. +
    3. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). If tasks fail independently with per-task residuals ri ≤ ε, then P(≥1 miss) = 1−∏i(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε: independence alone does not give equality. Tasks done by one model on one repository need not be independent at all. [corrected 2026-09-26]
    4. Instruction layer alone (q = 0): r = 1−p, so reaching ε needs p ≥ 1−ε from instructions. Raising p does bend the curve: for 30 independent tasks, P(≥1 miss) is 0.958 at p = 0.9 and 0.260 at p = 0.99.
    5. Deterministic layer alone (the bare model's p0): reaching ε needs q ≥ 1 − ε/(1−p0).
    6. Composition. Adding a check with P(it fires | M, no earlier check fired) > 0 strictly lowers r. Adding a copy of a check that is already present lowers nothing.
    -

    The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they can reach (1−pmax)(1−qmax). So a target ε with (1−pmax)(1−qmax) ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. Whether a real target falls in that range is an empirical question about p0, pmax and qmax, which this paper does not measure.

    +

    The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they reach r* = min(p,q)∈F (1−p)(1−q) over the joint feasible set F = {(p(π), q(π)) : π an admissible policy}. Instructions change which misses remain, and q is a catch rate conditional on that changed miss population, so pmax and qmax need not be attainable under one policy: (1−pmax)(1−qmax) is a lower bound on r*, reached only if the two maxima are jointly attainable on the same task distribution. For example, policy A with (p, q) = (0.5, 0.9) leaves 0.05 and policy B with (0.9, 0.1) leaves 0.09; the separate maxima, 0.9 and 0.9, suggest 0.01, which neither policy attains (research/recompute_corrections.py asserts this in §3b). So a target ε with r* ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. [corrected 2026-09-26] Whether a real target falls in that range is an empirical question about p0, pmax, qmax and the shape of F, which this paper does not measure.

    □

    @@ -267,7 +274,7 @@

    4 The central result — the two-layer duality theorem
    The two-layer duality architecture -
    Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Neither layer alone suffices — the formal content of the discipline never trust the output; earn trust with a check.
    +
    Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Where each factor is bounded away from zero, neither layer alone reaches a small residual [corrected 2026-09-26] — the formal content of the discipline never trust the output; earn trust with a check.
    @@ -295,6 +302,9 @@

    The honest cost side

    +

    5.4 A caught miss is not a completed task [corrected 2026-09-26]

    +

    Theorem D counts silent misses. A miss that a check catches is not thereby a completed, correct task: the gate can block the same turn repeatedly, the agent can abandon the task, and the repair can fail. So a lower silent-miss probability is not automatically a higher completed-correct-task rate, and the value of the two layers has to be measured on outcomes rather than read off r. The quantities that decide it are the true catch rate, the false-block rate, repaired success conditional on a catch, abandonment, added latency and recovery cost. This paper measures none of them.

    +

    6 The unified algorithm set — the TASK loop

    The faculties of Def. 4 are realized by seven algorithms. They are the reliability framework's A1–A7, recast here as the operations of the substrate: each is a faculty made mechanical, each binds to one lifecycle point, and together they form a single loop whose progress is guaranteed by an explicit worklist and whose floor is guaranteed by a deterministic gate.

    @@ -613,7 +623,7 @@

    12 Honest limits — what no architecture can guarantee

    13 Conclusion

    -

    A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the faculties it structurally lacks. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

    +

    A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the persistence and external checks it does not provide on its own [corrected 2026-09-26]. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

    That theorem is the formal content of a plain discipline: the output of a probability engine is never to be trusted on its own; trust is earned by an external check. The Qur'anic lens gives that discipline its oldest names — lā taqfu, do not pursue what you do not know; tabayyun, verify the report before you act; al-amāna, the weight of a trust accepted by one who may err. The mathematics says how to build the check. The tradition says why it is owed. The codebase shows it runs.

    Companion artifacts: the three-way crosswalk (JSON + markdown), the graded reference set (Appendix A), and two runnable prototype packages (impact-oracle, router-gate). This synthesis consolidates and does not supersede the v2 Theory → Evidence → Build-Map edition, which carries the empirical evidence layer and the full ecosystem map.

    @@ -742,8 +752,20 @@

    Corrections (2026-09-21)

  • The prototype results in §10 were refuted before this correction, and the section now says so. The impact oracle's “perfect recall” was measured on five mutations of a ten-file package the authors wrote; on 759 evaluated files in nine real repositories its recall was 0.022, and a grep baseline scored F1 0.437 against its 0.042. The router's “62.1% real cost saved” was measured on the 30 tasks its thresholds were tuned on; on 80 held-out tasks the pipeline's total spend was 20.2% higher than always using the premium tier. Per output the judge accepted, it cost $1.06 against always-premium's $1.76, but only 6 and 3 of 64 outputs were accepted, so neither figure is stable. The details are in the extended preprint and the refutation paper.
  • +

    Corrections (2026-09-26)

    +

    A second external deep review of the forgekit repository (2026-09-26, pinned at commit d2abfa69fb77531199ffc67c5c076b524af69040) re-checked the corrected Theorem D and this paper's framing. Each change is made in place above, marked [corrected 2026-09-26], and listed here with the original wording, so nothing is silently rewritten. The counterexample in item 1, the equality condition in item 2 and the 400-fold correction in item 4 are asserted by python3 research/recompute_corrections.py --theorem-checks (§3b), which needs no data. The PDF edition predates these corrections as well.

    +
      +
    1. Separately maximal rates need not be jointly attainable (§4). The range statement read: “together they can reach (1−pmax)(1−qmax)”. That requires both maxima to be attainable under one policy on the same task distribution; instructions change which misses remain, and q is conditional on that population. The attainable residual is min(p,q)∈F (1−p)(1−q) over F = {(p(π), q(π)) : admissible π}, and the product of the separate maxima is only a lower bound on it unless compatibility is established. Counterexample: policy A (0.5, 0.9) leaves 0.05, policy B (0.9, 0.1) leaves 0.09, and the separate maxima suggest 0.01, which neither attains.
    2. +
    3. Equality in the independent-task bound (§4, Claim 1). The text said P(≥1 miss) “equals 1−(1−ε)n only if tasks fail independently”. From per-task residuals ri ≤ ε, independence gives 1−∏(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε. With residuals 0.01, 0.005 and 0.001, for example, the probability is 0.0159 against the bound's 0.0297.
    4. +
    5. A caught miss is not a completed task (new §5.4). A lower silent-miss probability does not by itself mean more completed, correct tasks: a caught mistake can end in an abort, repeated blocking or a failed repair. The measures that decide the question — true catch rate, false-block rate, repaired success conditional on a catch, abandonment, latency and recovery cost — are listed and are not measured here.
    6. +
    7. Three lifecycle copies of one classifier are one detector (§5.3). The 2026-09-21 correction stands: multiplying identical checks at the Stop hook, pre-commit and CI as if independent understated the residual 400-fold. Copies at several points can widen the opportunities to run a check, such as edits made after a turn ended, but they are not independent semantic detectors.
    8. +
    9. The frozen-map premise, stated as the guarantees it removes (abstract, §2). The paper said “none of these can be fixed by prompting harder or by fine-tuning alone” and “any learning must be external”. Frozen parameters exclude weight updates during use; they do not exclude changed behaviour from examples, retrieved facts, feedback or additional computation in the current context, which GPT-3's few-shot evaluation measured with no gradient updates (Brown et al., 2020, arXiv:2005.14165). The missing guarantees are narrower: no built-in durable state across independent invocations; a bounded context; no automatic parameter update; and unreliable self-verification without external evidence. The substrate is one tested way of supplying persistence and verification, not the only logically possible architecture.
    10. +
    11. Prior art and what is (and is not) claimed (byline, abstract, §1, §13). External memory, feedback-driven improvement and structured agent control have clear prior art: CoALA (Sumers et al., 2023, arXiv:2309.02427) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023, arXiv:2303.11366) improves agents with linguistic feedback and an episodic memory buffer and no weight updates. Both were already among this paper's references. The defensible claim is a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes; novelty is claimed only for a specific protocol, invariant, evaluation result or integration that survives an explicit prior-art comparison. The five faculties are a decomposition, not a proof of necessity, and the conclusion no longer says the architecture “supplies the faculties it structurally lacks”. The byline still called the three bodies of work “independently-developed” after the 2026-09-21 corrections had shown that they share an author; it now says so.
    12. +
    13. Reference grades are bibliographic (Appendix A). confirmed and traceable record that a source exists and is correctly attributed. They are not a judgement that the source supports the theory it is cited for. graded_reference_set.md now keeps bibliographic verification, claim support, study design, independent replication and transfer scope as separate grades.
    14. +
    +

    Appendix A — Graded reference set (new sources)

    -

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21]

    +

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21] These grades are bibliographic: they confirm that a record exists and is correctly attributed. They do not grade whether a source supports the claim it is cited for, its study design, independent replication or transfer scope; research/formal-synthesis/graded_reference_set.md now keeps those as separate fields. [corrected 2026-09-26]

    SourceIDGradeNote
    Cognitive Architectures for Language Agents
    Theodore R. Sumers, Shunyu Yao, Karthik Narasi, 2023
    2309.02427confirmedRetrieved via arXiv metadata API; title/authors match claim exactly. Unifies memory, planning/reasoning, action, and learning modules into a single CoALA framework for language agents, giving the cognitive-substrate work's memory/im…
    diff --git a/research/python-prototypes/impact_oracle/README.md b/research/python-prototypes/impact_oracle/README.md index c3939095..3f66ba5a 100644 --- a/research/python-prototypes/impact_oracle/README.md +++ b/research/python-prototypes/impact_oracle/README.md @@ -56,8 +56,16 @@ in `oracle.py` as the module defaults: dependencies. Two terminal relations were added: `sibling` (one bounded forward hop to a bridge, then one bounded reverse hop from it, skipping bridges whose in-degree exceeds the cap) and `forward` (the changed symbol's own dependencies, ≤2 hops). - Held-out (never-tuned) repos at threshold 0.10: precision 0.320, recall 0.647, - **F1 0.428 vs the grep baseline's 0.371** — a reversal of the as-shipped 0.042 vs 0.437. + On the three held-out (never-tuned) repositories at the pre-registered canonical threshold + 0.02: precision 0.305, recall 0.653, **F1 0.416 against the grep baseline's 0.371** (ΔF1 about + +0.044). At threshold 0.10, which was selected on the tuning repositories, F1 is 0.428 + (precision 0.320, recall 0.647; ΔF1 +0.0565), and that is the only threshold with + per-repository counts: all three held-out repositories favour the repair there, but three of + three is a one-sided sign-test p of 0.125 and pytest supplies 71.3% of the held-out pairs, so + that it *beats* grep is not established. Never compare a 0.10 figure with a 0.02 one. (Corrected + 2026-09-26: this line quoted only the 0.10 result, as "F1 0.428 vs the grep baseline's 0.371 — a + reversal of the as-shipped 0.042 vs 0.437", which also set three held-out repositories against + all nine.) `ImpactOracle(wm, sibling_enabled=False, forward_enabled=False)` reproduces the as-shipped reverse-only traversal exactly; the untouched as-shipped package is archived as @@ -119,12 +127,16 @@ On this demo package the oracle reached recall 1.000 (it missed no affected modu these five mutations), with its best F1 of 0.79 at the optimal threshold (t=0.4). > **Refuted on real code.** That recall did not transfer. On 759 files in nine open-source -> Python repositories, with co-change ground truth, this version's recall was **0.022** and a +> Python repositories, with co-change ground truth, the as-shipped version's recall was **0.022** and a > grep baseline scored F1 0.437 against its 0.042: the traversal walks only reverse edges, and a -> construction defect breaks `src/`-layout packages. A repaired version ships in -> [`../../empirical-refutation/replication_package.tar.gz`](../../empirical-refutation/). Earlier -> versions of this README said the oracle "achieves perfect recall (never misses a truly affected -> module)". (Corrected 2026-09-21.) +> construction defect breaks `src/`-layout packages. This package **is** the repaired version (see +> "Repaired (v2)" above); the untouched as-shipped v1 is archived in +> [`../../empirical-refutation/replication_package.tar.gz`](../../empirical-refutation/), and +> `ImpactOracle(wm, sibling_enabled=False, forward_enabled=False)` reproduces its traversal. The +> results table above is the original five-mutation demonstration reported in the white paper (§8). Earlier versions of +> this README said the oracle "achieves perfect recall (never misses a truly affected module)". +> (Corrected 2026-09-21; the pointer to the repaired version corrected 2026-09-26 — it said "A +> repaired version ships in" the replication tarball.) ## File structure diff --git a/research/python-prototypes/router_gate/README.md b/research/python-prototypes/router_gate/README.md index 80cbd0df..0fda151f 100644 --- a/research/python-prototypes/router_gate/README.md +++ b/research/python-prototypes/router_gate/README.md @@ -173,3 +173,25 @@ python -m pytest ``` The included evaluation task set is a demonstration set, not a field benchmark. Calibrate thresholds on your own workload before enforcing automated model selection in high-stakes production paths. + +## Held-out result, and what "success" means + +*(Added 2026-09-26.)* The 30-task demonstration above was the set the thresholds were tuned on. On +80 pre-registered held-out tasks from real GitHub issues and pull requests +([`research/empirical-refutation/`](../../empirical-refutation/)), gate F1 was 0.37, and on the 64 +non-halted tasks the routed pipeline spent **$6.3582** against always-premium's **$5.2893** — +**20.21% more**, not saved. + +"Success" in that evaluation is **`judge_accepted`**: a model judge (the same model the pipeline +used as its mid-tier executor) accepted 6 of 64 routed outputs and 3 of 64 always-premium outputs. +That gives $1.060 against $1.763 per judge-accepted output, a ratio that is unstable at those counts. +No held-out task admitted execution-based verification, so **`tests_passed`**, **`human_accepted`** +and **`deployed_without_revert`** were never measured. Re-labelling 30 tasks with the same model and a +reworded prompt gave halt κ 0.5161 and tier κ 0.8919 — self-consistency, not agreement with a human. +The figures are recomputed from the replication archive by +[`research/recompute_corrections.py`](../../recompute_corrections.py) (§7, §9). + +When you evaluate this pipeline on your own workload, record those four outcomes separately, anchor +coding success on executable tests and blind human adjudication of disagreements, and keep gate +precision/recall, router solve rate and total pipeline cost as separate endpoints; count +clarification turns and tasks the gate rejected that were in fact valid. diff --git a/research/recompute_corrections.py b/research/recompute_corrections.py index 8d2c6ba8..4b89c248 100644 --- a/research/recompute_corrections.py +++ b/research/recompute_corrections.py @@ -8,11 +8,16 @@ python research/recompute_corrections.py rp/repro Sections 1-3 need no data (they are arithmetic on the synthesis paper's own worked -examples). Sections 4-9 read `results/*.json` and `data/*.parquet` from the package. +examples). Section 3b (added 2026-09-26) holds executable sanity checks on Theorem D that +need no data either; each is an `assert`, so a wrong statement fails the run with a +non-zero exit. Sections 4-9 read `results/*.json` and `data/*.parquet` from the package. Every random draw uses a fixed seed that is printed next to its result. -The corrections these numbers support were prompted by an external deep review of the -repository (2026-09-21); see the "Corrections" sections of each paper. + python research/recompute_corrections.py --theorem-checks # sections 1-3b only, no data + python research/recompute_corrections.py --help + +The corrections these numbers support were prompted by external deep reviews of the +repository (2026-09-21 and 2026-09-26); see the "Corrections" sections of each paper. """ import json @@ -117,6 +122,56 @@ def theorem_d(): print(f" T4: 150 lines x 80 bytes = {150 * 80} bytes > 8 KB cap (8192); 8192/150 = {8192 / 150:.1f} bytes/line") +def theorem_checks(): + """Executable sanity checks on Theorem D (2026-09-26 corrections). Each one is an assert.""" + header("3b. Theorem D sanity checks (asserted; 2026-09-26 corrections)") + tol = 1e-12 + + # (a) Separately maximal p and q need not be jointly attainable under one policy. The + # residual of a policy pi is (1 - p(pi)) * (1 - q(pi)); the target is its minimum over the + # joint feasible set F = {(p(pi), q(pi)) : admissible pi}, not the product of the maxima. + policies = {"A": (0.5, 0.9), "B": (0.9, 0.1)} + residual = {name: (1 - p) * (1 - q) for name, (p, q) in policies.items()} + p_max = max(p for p, _ in policies.values()) + q_max = max(q for _, q in policies.values()) + naive = (1 - p_max) * (1 - q_max) + attained = min(residual.values()) + for name, (p, q) in policies.items(): + print(f" policy {name}: (p, q) = ({p}, {q}) -> residual (1-p)(1-q) = {residual[name]:.4f}") + print(f" separate maxima p_max = {p_max}, q_max = {q_max} -> (1-p_max)(1-q_max) = {naive:.4f}") + print(f" lowest residual any feasible policy attains = {attained:.4f}") + assert abs(residual["A"] - 0.05) < tol, residual["A"] + assert abs(residual["B"] - 0.09) < tol, residual["B"] + assert abs(naive - 0.01) < tol, naive + assert naive < attained, "the product of separate maxima is not attained by any policy here" + print(" => (1-p_max)(1-q_max) is a LOWER bound on the attainable residual unless the maxima are") + print(" jointly attainable; optimise min over F = {(p(pi), q(pi))} of (1-p)(1-q) instead") + + # (b) Over n independent tasks with per-task residuals r_i <= eps, + # P(>=1 miss) = 1 - prod(1 - r_i) <= 1 - (1 - eps)^n, with EQUALITY only when every r_i = eps. + # Independence alone does not give equality; the union bound n * eps needs neither. + eps = 0.01 + unequal = [0.01, 0.005, 0.001] + n = len(unequal) + bound = 1 - (1 - eps) ** n + p_unequal = 1 - math.prod(1 - r for r in unequal) + p_equal = 1 - math.prod(1 - r for r in [eps] * n) + print(f" n={n}, eps={eps}: 1-(1-eps)^n = {bound:.6f}; union bound n*eps = {n * eps:.6f}") + print(f" residuals {unequal}: 1-prod(1-r_i) = {p_unequal:.6f} (strictly below the bound)") + print(f" residuals all equal to eps: 1-prod(1-r_i) = {p_equal:.6f} (equality)") + assert p_unequal < bound - tol, (p_unequal, bound) + assert abs(p_equal - bound) < tol, (p_equal, bound) + assert bound <= n * eps + tol, (bound, n * eps) + + # (c) Three lifecycle copies of one classifier are not three independent detectors: the + # independence product understates the nested residual (1-p)(1-c) 400-fold (section 2). + p, c, k = 0.7, 0.95, 3 + ratio = ((1 - p) * (1 - c)) / ((1 - p) * (1 - c) ** k) + print(f" {k} copies of one check (p={p}, c={c}): nested / independence-product residual = {ratio:.0f}x") + assert round(ratio) == 400, ratio + print(" all Theorem D sanity checks passed") + + # -------------------------------------------------------------------------------------- # Minimal Parquet reader (flat schema; PLAIN/dictionary; UNCOMPRESSED or SNAPPY) # -------------------------------------------------------------------------------------- @@ -344,6 +399,12 @@ def pooled(rows, idx): po, pg = pooled(oracle, full), pooled(grep, full) print(f" pooled oracle P/R/F1 = {po[0]:.4f} / {po[1]:.4f} / {po[2]:.4f}") print(f" pooled grep P/R/F1 = {pg[0]:.4f} / {pg[1]:.4f} / {pg[2]:.4f}") + # Repository-level view (added 2026-09-26): macro F1 weights each repository equally, so one + # large repository cannot carry the pooled figure. F1 is 0 where a method predicts nothing. + f1o = [prf(*oracle[i])[2] for i in full] + f1g = [prf(*grep[i])[2] for i in full] + print(f" macro F1 over {len(repos)} repositories: oracle {sum(f1o) / len(f1o):.4f}, grep {sum(f1g) / len(f1g):.4f}") + print(f" repositories where grep F1 > oracle F1: {sum(g > o for o, g in zip(f1o, f1g))} of {len(repos)}") rng = random.Random(SEED) draws = [] for _ in range(B): @@ -484,11 +545,18 @@ def w(i, j): def main(): + args = sys.argv[1:] + if args and args[0] in ("-h", "--help"): + print(__doc__) + return theorem_d() - if len(sys.argv) < 2: + theorem_checks() + if args and args[0] == "--theorem-checks": + return + if not args: print("\n(pass the extracted replication package's repro/ directory to recompute sections 4-9)") return - pkg = sys.argv[1] + pkg = args[0] ground_truth(pkg) cluster_bootstrap(pkg) repaired_vs_grep(pkg) diff --git a/scripts/build-pages.mjs b/scripts/build-pages.mjs index 4e774a87..c274a93d 100644 --- a/scripts/build-pages.mjs +++ b/scripts/build-pages.mjs @@ -237,7 +237,7 @@ h2{font-size:var(--fs-2);letter-spacing:-.01em;margin:0 0 var(--sp-4)} code{font:500 var(--fs-n1) var(--mono);background:var(--panel-2);border:1px solid var(--line);border-radius:var(--r-s);padding:0 var(--sp-2)} footer{padding:var(--sp-8) 0;color:var(--faint);font-size:var(--fs-n1)} @media(max-width:800px){.grid{grid-template-columns:1fr}} -

    ${esc(d.name)} · v${esc(d.version)} · Node ${esc(d.node)}

    Live status, straight from the repository.

    ${esc(d.description)}

    Install in 60 seconds Read the docs

    ${esc(d.license)} license${esc(d.deps)} runtime dependencies${esc(d.branch)} @ ${esc(d.commit)}${live}
    ${esc(d.impact)}
    blast-radius lookup

    Measured from this repo's benchmark report, not a marketing placeholder.

    reports/benchmarks.md

    ${esc(d.speed)}
    pre-action gate

    Assumptions, routing, reuse, context, impact, scope, and anchoring.

    reports/benchmarks.md

    ${esc(d.saved.match(/^[\d.]+\s*%?/)?.[0] ?? d.saved)}
    ${esc(d.saved.replace(/^[\d.]+\s*%?\s*/, "") || "routing signal")}

    Documented from the white-paper prototype and exposed by Forge cost reports.

    whitepaper prototype

    Quickstart

    npm install -g @codewithjuber/forgekit +

    ${esc(d.name)} · v${esc(d.version)} · Node ${esc(d.node)}

    Live status, straight from the repository.

    ${esc(d.description)}

    Install in 60 seconds Read the docs

    ${esc(d.license)} license${esc(d.deps)} runtime dependencies${esc(d.branch)} @ ${esc(d.commit)}${live}
    ${esc(d.impact)}
    blast-radius lookup

    Measured from this repo's benchmark report, not a marketing placeholder.

    reports/benchmarks.md

    ${esc(d.speed)}
    pre-action gate

    Assumptions, routing, reuse, context, impact, scope, and anchoring.

    reports/benchmarks.md

    ${esc(d.saved.match(/^[\d.]+\s*%?/)?.[0] ?? d.saved)}
    ${esc(d.saved.replace(/^[\d.]+\s*%?\s*/, "") || "routing signal")}

    Held-out result: on 80 pre-registered tasks the white-paper router spent this much more than always-premium; its 62.1% demo saving is refuted.

    research/empirical-refutation

    Quickstart

    npm install -g @codewithjuber/forgekit forge init forge doctor forge substrate "Change auth validation and update tests"

    Latest repo changes

      ${d.latest.map((x) => `
    • ${esc(x)}
    • `).join("")}

    Benchmark sections indexed: ${esc(d.benchMentions)} · benchmarks file updated ${esc(d.benchUpdated)}.

    Data Sources

    No mock data is used. This page is regenerated from repository files during CI (generated ${esc(d.generated)} from ${esc(d.commit)}). Enable BUILD_PAGES_LIVE=1 to refresh public GitHub counters with ETag/Last-Modified caching.

    • package.json
    • README.md
    • CHANGELOG.md
    • reports/benchmarks.md
    • ${api} (optional, no auth, only when BUILD_PAGES_LIVE=1)
    WCAG-minded semantic HTML, keyboard focus, responsive 320px–1920px+, and reduced-motion-safe. Same color and font tokens as the landing page — parity enforced in test/pages.test.js.
    `; diff --git a/scripts/claims-status.mjs b/scripts/claims-status.mjs new file mode 100644 index 00000000..24d0c8c1 --- /dev/null +++ b/scripts/claims-status.mjs @@ -0,0 +1,340 @@ +#!/usr/bin/env node +/** + * The claim/status registry, checked and rendered (node stdlib only). + * + * `docs/status/claims.json` records every load-bearing headline the project makes — what is + * claimed, which component and version it is about, the commit it was assessed against, its + * status, and the evidence behind it. This script keeps three things honest: + * + * 1. the registry itself: required fields, a closed set of statuses, unique ids, a commit + * that looks like one, and evidence paths that exist in the repository; + * 2. the status table in `docs/status/README.md`, which is generated from the registry + * between the CLAIMS:BEGIN / CLAIMS:END markers and never edited by hand; + * 3. the research copies under `docs/cognitive-substrate/`, which must stay byte-identical + * (sha256) to their canonical sources under `research/cognitive-substrate/`. + * + * Usage: + * node scripts/claims-status.mjs # validate, rewrite the table, check copies + * node scripts/claims-status.mjs --check # validate + fail on any drift (CI mode) + * node scripts/claims-status.mjs --sync-copies # re-copy canonical research files first + * node scripts/claims-status.mjs --root # operate on another checkout + * + * Exit codes: 0 ok · 1 invalid registry or drift · 2 usage error. + */ +import { createHash } from "node:crypto"; +import { copyFileSync, existsSync, readFileSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +/** The closed set of statuses a claim may carry, in display order. */ +export const STATUSES = ["implemented", "measured", "partial", "reported", "hypothesis", "refuted"]; + +/** Every claim must carry each of these fields. */ +export const REQUIRED_FIELDS = [ + "id", + "claim", + "component", + "version", + "source_commit", + "status", + "evidence", + "notes", +]; + +export const REGISTRY_PATH = "docs/status/claims.json"; +export const README_PATH = "docs/status/README.md"; +export const BEGIN = ""; +export const END = ""; + +/** + * Byte-identical copies kept under docs/ for readers there: [copy, canonical source]. + * `research/` is canonical; `--sync-copies` copies source → copy. + * @type {ReadonlyArray} + */ +export const COPY_PAIRS = [ + [ + "docs/cognitive-substrate/cognitive_substrate_whitepaper.html", + "research/cognitive-substrate/cognitive_substrate_whitepaper.html", + ], + [ + "docs/cognitive-substrate/cognitive_substrate_whitepaper.pdf", + "research/cognitive-substrate/cognitive_substrate_whitepaper.pdf", + ], + [ + "docs/cognitive-substrate/evidence_map.md", + "research/cognitive-substrate/evidence/evidence_map.md", + ], + [ + "docs/cognitive-substrate/ecosystem_map.md", + "research/cognitive-substrate/evidence/ecosystem_map.md", + ], +]; + +const ID_RE = /^[a-z0-9][a-z0-9-]*$/; +const COMMIT_RE = /^[0-9a-f]{7,40}$/; +const URL_RE = /^https?:\/\//; + +const DEFAULT_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); + +/** + * Validate a parsed registry. Pure apart from optional evidence-path existence checks, which + * run only when `root` is given. + * @param {any} registry the parsed claims.json + * @param {{root?: string|null}} [opts] + * @returns {string[]} human-readable problems; empty means valid + */ +export function validateRegistry(registry, { root = null } = {}) { + const errors = []; + if (!registry || typeof registry !== "object" || Array.isArray(registry)) { + return ["registry must be a JSON object with a `claims` array"]; + } + if (!Array.isArray(registry.claims) || registry.claims.length === 0) { + return ["registry.claims must be a non-empty array"]; + } + const seen = new Set(); + registry.claims.forEach((c, i) => { + const where = `claims[${i}]${c && typeof c.id === "string" ? ` (${c.id})` : ""}`; + if (!c || typeof c !== "object" || Array.isArray(c)) { + errors.push(`${where}: must be an object`); + return; + } + for (const f of REQUIRED_FIELDS) { + if (!(f in c)) errors.push(`${where}: missing required field "${f}"`); + } + for (const f of ["id", "claim", "component", "version", "source_commit", "status"]) { + if (f in c && (typeof c[f] !== "string" || !c[f].trim())) { + errors.push(`${where}: "${f}" must be a non-empty string`); + } + } + if ("notes" in c && typeof c.notes !== "string") { + errors.push(`${where}: "notes" must be a string`); + } + if (typeof c.id === "string" && c.id) { + if (!ID_RE.test(c.id)) errors.push(`${where}: id must match ${ID_RE}`); + if (seen.has(c.id)) errors.push(`${where}: duplicate id "${c.id}"`); + seen.add(c.id); + } + if (typeof c.status === "string" && c.status && !STATUSES.includes(c.status)) { + errors.push(`${where}: status "${c.status}" is not one of ${STATUSES.join(", ")}`); + } + if (typeof c.source_commit === "string" && c.source_commit) { + if (!COMMIT_RE.test(c.source_commit)) { + errors.push(`${where}: source_commit must be 7-40 lowercase hex characters`); + } + } + if ("evidence" in c) { + if (!Array.isArray(c.evidence) || c.evidence.length === 0) { + errors.push(`${where}: evidence must be a non-empty array of paths or URLs`); + } else { + for (const e of c.evidence) { + if (typeof e !== "string" || !e.trim()) { + errors.push(`${where}: every evidence entry must be a non-empty string`); + continue; + } + if (root && !URL_RE.test(e)) { + const rel = e.split("#")[0]; + if (!existsSync(path.join(root, rel))) { + errors.push(`${where}: evidence path not found: ${rel}`); + } + } + } + } + } + }); + return errors; +} + +/** Escape a value for a single Markdown table cell. */ +const cell = (s) => + String(s ?? "") + .replace(/\r?\n/g, " ") + .replace(/\|/g, "\\|") + .trim(); + +/** + * One evidence entry as a Markdown link: URLs keep their address; repository paths become + * links relative to docs/status/README.md. + * @param {string} e + */ +export function evidenceLink(e) { + if (URL_RE.test(e)) { + const label = e.replace(URL_RE, "").replace(/\/$/, ""); + return `[${cell(label)}](${e})`; + } + return `[\`${cell(e)}\`](../../${e})`; +} + +/** + * Render the generated block (without the markers): a status summary and one table row per + * claim, in registry order. + * @param {any} registry a valid registry + * @returns {string} + */ +export function renderTable(registry) { + const claims = registry.claims; + const counts = STATUSES.map((s) => [s, claims.filter((c) => c.status === s).length]); + const commits = [...new Set(claims.map((c) => c.source_commit))]; + const lines = [ + `${claims.length} claims — ${counts.map(([s, n]) => `${s} ${n}`).join(" · ")}.`, + `Assessed against commit${commits.length > 1 ? "s" : ""} ${commits.map((c) => `\`${c.slice(0, 12)}\``).join(", ")}${registry.as_of ? ` (as of ${registry.as_of})` : ""}.`, + "", + "| ID | Status | Claim | Component · version | Evidence | Notes |", + "| --- | --- | --- | --- | --- | --- |", + ]; + for (const c of claims) { + lines.push( + `| \`${cell(c.id)}\` | **${cell(c.status)}** | ${cell(c.claim)} | ${cell(c.component)} · ${cell(c.version)} | ${c.evidence.map(evidenceLink).join("
    ")} | ${cell(c.notes)} |`, + ); + } + return lines.join("\n"); +} + +/** + * Replace the generated block in the README text. + * @param {string} text current README + * @param {string} block rendered block (no markers) + * @returns {string} + */ +export function spliceReadme(text, block) { + const begin = text.indexOf(BEGIN); + const end = text.indexOf(END); + if (begin === -1 || end === -1 || end < begin) { + throw new Error(`${README_PATH} must contain the markers ${BEGIN} … ${END}`); + } + return `${text.slice(0, begin)}${BEGIN}\n${block}\n${text.slice(end)}`; +} + +/** @param {string} file */ +export const sha256File = (file) => createHash("sha256").update(readFileSync(file)).digest("hex"); + +/** + * Compare each docs copy with its canonical source. + * @param {string} root + * @param {ReadonlyArray} [pairs] + * @returns {{copy: string, source: string, ok: boolean, reason: string}[]} + */ +export function checkCopies(root, pairs = COPY_PAIRS) { + return pairs.map(([copy, source]) => { + const c = path.join(root, copy); + const s = path.join(root, source); + if (!existsSync(s)) return { copy, source, ok: false, reason: "canonical source missing" }; + if (!existsSync(c)) return { copy, source, ok: false, reason: "copy missing" }; + const same = sha256File(c) === sha256File(s); + return { copy, source, ok: same, reason: same ? "identical" : "sha256 differs" }; + }); +} + +/** + * Copy every drifted canonical source over its docs copy. + * @param {string} root + * @param {ReadonlyArray} [pairs] + * @returns {string[]} the copies rewritten + */ +export function syncCopies(root, pairs = COPY_PAIRS) { + const written = []; + for (const r of checkCopies(root, pairs)) { + if (r.ok || r.reason === "canonical source missing") continue; + copyFileSync(path.join(root, r.source), path.join(root, r.copy)); + written.push(r.copy); + } + return written; +} + +/** + * The CLI. Returns an exit code instead of exiting, so tests can drive it. + * @param {string[]} argv + * @param {{root?: string, pairs?: ReadonlyArray, + * log?: (s: string) => void, error?: (s: string) => void}} [io] + * @returns {number} + */ +export function run(argv, io = {}) { + const log = io.log ?? ((s) => process.stdout.write(`${s}\n`)); + const error = io.error ?? ((s) => process.stderr.write(`${s}\n`)); + let root = io.root ?? DEFAULT_ROOT; + const pairs = io.pairs ?? COPY_PAIRS; + const flags = new Set(); + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (a === "--root") { + const next = argv[++i]; + if (!next) { + error("--root needs a directory"); + return 2; + } + root = path.resolve(next); + } else if (a === "--check" || a === "--sync-copies") flags.add(a); + else if (a === "--help" || a === "-h") { + log( + "usage: node scripts/claims-status.mjs [--check] [--sync-copies] [--root ]\n" + + " (no flag) validate the registry, rewrite the table in docs/status/README.md, check copies\n" + + " --check validate and fail (exit 1) on a stale table or a drifted docs copy\n" + + " --sync-copies copy canonical research/ files over drifted docs/ copies first", + ); + return 0; + } else { + error(`unknown argument: ${a}`); + return 2; + } + } + const check = flags.has("--check"); + + let registry; + try { + registry = JSON.parse(readFileSync(path.join(root, REGISTRY_PATH), "utf8")); + } catch (e) { + error(`cannot read ${REGISTRY_PATH}: ${/** @type {Error} */ (e).message}`); + return 1; + } + const problems = validateRegistry(registry, { root }); + if (problems.length) { + for (const p of problems) error(`registry: ${p}`); + return 1; + } + + let failed = false; + const readmeFile = path.join(root, README_PATH); + let current; + try { + // Compare line content, not line endings: a Windows checkout (core.autocrlf) has CRLF. + current = readFileSync(readmeFile, "utf8").replace(/\r\n/g, "\n"); + } catch (e) { + error(`cannot read ${README_PATH}: ${/** @type {Error} */ (e).message}`); + return 1; + } + let expected; + try { + expected = spliceReadme(current, renderTable(registry)); + } catch (e) { + error(/** @type {Error} */ (e).message); + return 1; + } + if (expected !== current) { + if (check) { + error(`${README_PATH} is stale — run \`node scripts/claims-status.mjs\` to regenerate it`); + failed = true; + } else { + writeFileSync(readmeFile, expected); + log(`rewrote the generated table in ${README_PATH}`); + } + } + + if (flags.has("--sync-copies") && !check) { + for (const c of syncCopies(root, pairs)) log(`re-copied ${c} from its canonical source`); + } + for (const r of checkCopies(root, pairs)) { + if (!r.ok) { + error(`copy drift: ${r.copy} vs ${r.source} — ${r.reason} (run with --sync-copies)`); + failed = true; + } + } + + if (failed) return 1; + log( + `ok: ${registry.claims.length} claims valid, ${README_PATH} current, ${pairs.length} docs copies identical`, + ); + return 0; +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + process.exit(run(process.argv.slice(2))); +} diff --git a/test/claims_status.test.js b/test/claims_status.test.js new file mode 100644 index 00000000..8ec60701 --- /dev/null +++ b/test/claims_status.test.js @@ -0,0 +1,211 @@ +// The claim/status registry (docs/status/claims.json) and its generator/checker +// (scripts/claims-status.mjs): validation rules, table rendering, --check drift detection on a +// throwaway checkout, and the repository's own registry staying valid and in sync. +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import { test } from "node:test"; +import { + BEGIN, + checkCopies, + END, + README_PATH, + REGISTRY_PATH, + renderTable, + run, + STATUSES, + spliceReadme, + validateRegistry, +} from "../scripts/claims-status.mjs"; + +const SHA = "d2abfa69fb77531199ffc67c5c076b524af69040"; + +/** A minimal valid claim; override any field. */ +const claim = (over = {}) => ({ + id: "example-claim", + claim: "The example does what it says.", + component: "example component", + version: "1.0.0", + source_commit: SHA, + status: "measured", + evidence: ["evidence/result.md"], + notes: "", + ...over, +}); + +/** Run the CLI against `root`, capturing its output instead of printing it. */ +function cli(argv, root, pairs) { + const out = []; + const err = []; + const code = run(argv, { + root, + pairs, + log: (s) => out.push(s), + error: (s) => err.push(s), + }); + return { code, out: out.join("\n"), err: err.join("\n") }; +} + +/** A throwaway checkout with a registry, a README carrying the markers, and one docs copy. */ +function fixture(claims = [claim()]) { + const root = mkdtempSync(join(tmpdir(), "forge-claims-")); + const write = (rel, text) => { + mkdirSync(dirname(join(root, rel)), { recursive: true }); + writeFileSync(join(root, rel), text); + }; + write("evidence/result.md", "# result\n"); + write(REGISTRY_PATH, `${JSON.stringify({ as_of: "2026-09-26", claims }, null, 2)}\n`); + write(README_PATH, `# Claim status\n\n${BEGIN}\n${END}\n`); + write("research/paper.html", "

    canonical

    \n"); + write("docs/paper.html", "

    canonical

    \n"); + const pairs = [["docs/paper.html", "research/paper.html"]]; + return { root, write, pairs, cleanup: () => rmSync(root, { recursive: true, force: true }) }; +} + +test("a valid registry has no problems", () => { + assert.deepEqual(validateRegistry({ claims: [claim()] }), []); + for (const status of STATUSES) { + assert.deepEqual(validateRegistry({ claims: [claim({ status })] }), [], status); + } +}); + +test("validation rejects a missing field, an unknown status and a duplicate id", () => { + const { notes: _drop, ...noNotes } = claim(); + const problems = validateRegistry({ + claims: [noNotes, claim({ id: "b", status: "proven" }), claim({ id: "b" })], + }); + assert.ok( + problems.some((p) => /missing required field "notes"/.test(p)), + problems.join("\n"), + ); + assert.ok( + problems.some((p) => /status "proven" is not one of/.test(p)), + problems.join("\n"), + ); + assert.ok( + problems.some((p) => /duplicate id "b"/.test(p)), + problems.join("\n"), + ); +}); + +test("validation rejects bad commits, empty evidence, malformed ids and empty registries", () => { + const problems = validateRegistry({ + claims: [ + claim({ id: "x", source_commit: "HEAD" }), + claim({ id: "y", evidence: [] }), + claim({ id: "Bad Id" }), + claim({ id: "z", claim: " " }), + ], + }); + assert.ok(problems.some((p) => /source_commit must be 7-40/.test(p))); + assert.ok(problems.some((p) => /evidence must be a non-empty array/.test(p))); + assert.ok(problems.some((p) => /id must match/.test(p))); + assert.ok(problems.some((p) => /"claim" must be a non-empty string/.test(p))); + assert.deepEqual(validateRegistry({ claims: [] }), ["registry.claims must be a non-empty array"]); + assert.deepEqual(validateRegistry([]), ["registry must be a JSON object with a `claims` array"]); +}); + +test("with a root, a missing evidence path is a problem and a URL is not checked", () => { + const f = fixture(); + try { + const reg = { + claims: [ + claim({ evidence: ["evidence/result.md#section", "https://example.org/paper"] }), + claim({ id: "gone", evidence: ["evidence/missing.md"] }), + ], + }; + const problems = validateRegistry(reg, { root: f.root }); + assert.deepEqual(problems, ["claims[1] (gone): evidence path not found: evidence/missing.md"]); + } finally { + f.cleanup(); + } +}); + +test("the rendered table escapes pipes, links evidence and counts statuses", () => { + const out = renderTable({ + as_of: "2026-09-26", + claims: [ + claim({ claim: "a | b", evidence: ["evidence/result.md", "https://example.org/x/"] }), + claim({ id: "second", status: "refuted" }), + ], + }); + assert.match(out, /^2 claims — implemented 0 · measured 1 · .*refuted 1\./); + assert.match(out, /Assessed against commit `d2abfa69fb77` \(as of 2026-09-26\)/); + assert.match(out, /a \\\| b/); + assert.match(out, /\[`evidence\/result\.md`\]\(\.\.\/\.\.\/evidence\/result\.md\)/); + assert.match(out, /\[example\.org\/x\]\(https:\/\/example\.org\/x\/\)/); + assert.throws(() => spliceReadme("no markers here", out), /must contain the markers/); +}); + +test("--check passes when current, and fails on a stale table without rewriting it", () => { + const f = fixture(); + try { + assert.equal(cli(["--check"], f.root, f.pairs).code, 1, "a fresh README has no table yet"); + const first = cli([], f.root, f.pairs); + assert.equal(first.code, 0, first.err); + assert.match(first.out, /rewrote the generated table/); + assert.equal(cli(["--check"], f.root, f.pairs).code, 0); + + // Change a status in the registry: the table on disk is now stale. + const reg = JSON.parse(readFileSync(join(f.root, REGISTRY_PATH), "utf8")); + reg.claims[0].status = "refuted"; + f.write(REGISTRY_PATH, JSON.stringify(reg)); + const before = readFileSync(join(f.root, README_PATH), "utf8"); + const stale = cli(["--check"], f.root, f.pairs); + assert.equal(stale.code, 1); + assert.match(stale.err, /is stale/); + assert.equal(readFileSync(join(f.root, README_PATH), "utf8"), before, "--check never writes"); + + assert.equal(cli([], f.root, f.pairs).code, 0); + assert.equal(cli(["--check"], f.root, f.pairs).code, 0); + } finally { + f.cleanup(); + } +}); + +test("--check fails on an invalid registry and on a drifted docs copy; --sync-copies repairs it", () => { + const f = fixture(); + try { + assert.equal(cli([], f.root, f.pairs).code, 0); + f.write("docs/paper.html", "

    edited in the wrong place

    \n"); + assert.deepEqual( + checkCopies(f.root, f.pairs).map((r) => r.reason), + ["sha256 differs"], + ); + const drift = cli(["--check"], f.root, f.pairs); + assert.equal(drift.code, 1); + assert.match(drift.err, /copy drift: docs\/paper\.html vs research\/paper\.html/); + + const synced = cli(["--sync-copies"], f.root, f.pairs); + assert.equal(synced.code, 0, synced.err); + assert.equal(readFileSync(join(f.root, "docs/paper.html"), "utf8"), "

    canonical

    \n"); + assert.equal(cli(["--check"], f.root, f.pairs).code, 0); + + f.write(REGISTRY_PATH, JSON.stringify({ claims: [claim({ status: "vibes" })] })); + const invalid = cli(["--check"], f.root, f.pairs); + assert.equal(invalid.code, 1); + assert.match(invalid.err, /status "vibes"/); + assert.equal(cli(["--nope"], f.root, f.pairs).code, 2); + } finally { + f.cleanup(); + } +}); + +test("--check reads a CRLF checkout (Windows core.autocrlf) as current", () => { + const f = fixture(); + try { + assert.equal(cli([], f.root, f.pairs).code, 0); + const lf = readFileSync(join(f.root, README_PATH), "utf8"); + f.write(README_PATH, lf.replace(/\n/g, "\r\n")); + const r = cli(["--check"], f.root, f.pairs); + assert.equal(r.code, 0, r.err); + } finally { + f.cleanup(); + } +}); + +test("the repository's own registry is valid, its table current, and its docs copies identical", () => { + const r = cli(["--check"]); + assert.equal(r.code, 0, r.err); +});
    SourceIDGradeNote
    Cognitive Architectures for Language Agents
    Theodore R. Sumers, Shunyu Yao, Karthik Narasi, 2023
    2309.02427confirmedRetrieved via arXiv metadata API; title/authors match claim exactly. Unifies memory, planning/reasoning, action, and learning modules into a single CoALA framework for language agents, giving the cognitive-substrate work's memory/im…