diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c95f8faa..66e76869 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,6 +1,7 @@ # CI gate for every push and PR: the Node 20/22 test matrix (matches the ">=20" engines -# field; 18 is EOL) plus the shared quality gate (Biome, typecheck, ShellCheck, zero-dep -# assertion, version-drift, docs-drift, pack). The quality gate is the SAME reusable workflow +# field; 18 is EOL), the research contracts (Python prototype suites + recomputation), plus +# the shared quality gate (Biome, typecheck, ShellCheck, zero-dep assertion, version-drift, +# docs-drift, pack). The quality gate is the SAME reusable workflow # the version bump and release require, so none of the three can drift from the others. name: CI @@ -58,6 +59,38 @@ jobs: awk '/^not ok /{p=1} p{print} p&&/^ \.\.\.[[:space:]]*$/{p=0}' /tmp/win-test.log | head -200 exit "$ec" + # The research contracts (review A02): the two Python prototypes' own suites, the Theorem-D + # sanity checks (no data needed), and the recomputation of every corrected number from the + # replication package shipped in this repo. A research claim that stops recomputing fails CI + # like a broken unit test. + research: + name: Research (Python prototypes + recomputation) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-python@v6 + with: + python-version: "3.12" + - name: Install prototype test dependencies + run: | + python -m pip install --upgrade pip + python -m pip install \ + -r research/python-prototypes/impact_oracle/requirements.txt \ + -r research/python-prototypes/router_gate/requirements.txt + - name: impact_oracle tests + working-directory: research/python-prototypes/impact_oracle + run: python -m pytest -q + - name: router_gate tests + working-directory: research/python-prototypes/router_gate + run: python -m pytest -q + - name: Theorem sanity checks (no data) + run: python research/recompute_corrections.py --theorem-checks + - name: Recompute the corrections from the replication package + run: | + mkdir -p "$RUNNER_TEMP/rp" + tar -xzf research/empirical-refutation/replication_package.tar.gz -C "$RUNNER_TEMP/rp" + python research/recompute_corrections.py "$RUNNER_TEMP/rp/repro" + quality-gate: name: Quality gate uses: ./.github/workflows/reusable-quality-gate.yml diff --git a/.github/workflows/reusable-quality-gate.yml b/.github/workflows/reusable-quality-gate.yml index f3cc0dc4..836a1c93 100644 --- a/.github/workflows/reusable-quality-gate.yml +++ b/.github/workflows/reusable-quality-gate.yml @@ -33,4 +33,6 @@ jobs: run: node -e "process.exit(Object.keys(require('./package.json').dependencies||{}).length)" - run: node scripts/bump.mjs check - run: node src/cli.js docs check + - name: Claim registry and research copies are current + run: node scripts/claims-status.mjs --check - run: npm pack --dry-run diff --git a/.gitignore b/.gitignore index 118cb3a7..a7448fe1 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,11 @@ node_modules/ npm-debug.log* *.tgz +# Python bytecode / test caches (research prototypes) +__pycache__/ +*.pyc +.pytest_cache/ + # Forge runtime artifacts (generated, never committed) .forge/ *.log diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index f7b281be..3c55a1ac 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -1,11 +1,13 @@ # forgekit — architecture -> **One brain for every AI coding agent.** A large language model is stateless: one -> context window, wiped every call. It has no memory of what your team learned, no -> foresight about what an edit will break, and no enforced guardrails. forgekit is the -> **cognitive substrate** — the layer that runs _before_ the model edits code, supplying -> proof-carrying memory, impact foresight, and enforced guardrails — and a **cross-tool -> config compiler** that delivers that brain as native config into every tool at once. +> **A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, +> and explicit verification around coding agents.** A language model keeps no durable state +> between independent calls and sees only a bounded context window, so on its own it does not +> carry what your team learned, cannot see the parts of the repository an edit affects unless +> they are in context, and cannot enforce rules on itself. forgekit is the **cognitive +> substrate** — the layer that runs _before_ the model edits code, supplying evidence-referenced +> ("proof-carrying") memory, heuristic impact analysis, and guardrails — and a **cross-tool +> config compiler** that delivers it as native config into every tool at once. This document is the architecture reference. It is organized around four diagrams: @@ -194,10 +196,13 @@ flowchart LR class SV accent; ``` -The completeness gate on the retrieval side is `forge context ""`: it assembles a -budgeted context via set-cover over the predicted edit set (`R(edit)`), applies a -compression ladder, and reports the _computed missing set_ — the inputs it could not -assemble. That missing set is exactly what the substrate pipeline's context stage reads +The completeness gate on the retrieval side is `forge context ""`: it pins the +required-knowledge set for the edit (`R(edit)`), downgrades items along a compression ladder +before dropping anything, fills the rest of the budget with a value-density heuristic (no +approximation guarantee), and reports the _computed missing set_ — the inputs it could not +assemble — plus pending reads and truncations. Its "complete" means syntactically delivered +within an estimated token budget, not semantically sufficient +([plan 04 §7](docs/plans/substrate-v2/04-context-assembly.md#7-status-2026-09-26--partial)). That missing set is exactly what the substrate pipeline's context stage reads to decide whether an edit is safe to start. Surface: `forge reuse query | mint | stats`. ## 5. The end-to-end reliability layer @@ -597,7 +602,7 @@ forgekit/ learn_consolidate.js # bin/learn-consolidate.sh: deterministic consolidation of ~/.claude/skills/learned — merge duplicates, drop only ledger-refuted (dormant/retracted/attic) lessons; no model call reuse.js # proof-carrying artifact cache: fingerprint (MinHash+LSH), exact→near→adapt→miss ladder, atlas revalidation embed.js # optional embeddings tier (ADR-0005): FORGE_EMBED=cmd:|http:, swaps MinHash/Jaccard for cosine in `reuse query`/`ledger query`, disk-cached at .forge/embed-cache.jsonl, silent fallback to MinHash - context.js # budgeted context assembly + completeness gate: R(edit) set cover, compression ladder, computed missing-set + context.js # budgeted context assembly + completeness gate: R(edit) coverage, compression ladder, computed missing-set diagnose.js # doom-loop diagnosis: normalized failure signatures; 3× = diagnosis claim + one-tier escalation imagine.js # consequence simulation (Eq. 4): predicted breaks + minimal dry-run suite via greedy set cover uifingerprint.js # deterministic design fingerprint + slop-distance / conformance gate (no LLM, no screenshots) @@ -670,24 +675,21 @@ from the tree it describes. ```mermaid %%{init: {'theme':'base','themeVariables':{'primaryColor':'#201a15','primaryTextColor':'#f2ede7','primaryBorderColor':'#372c22','lineColor':'#f26430','secondaryColor':'#272019','tertiaryColor':'#171310','edgeLabelBackground':'#201a15','clusterBkg':'#171310','clusterBorder':'#4a3b2e','fontFamily':'ui-sans-serif, system-ui, sans-serif','fontSize':'14px'},'flowchart':{'curve':'basis','padding':10,'nodeSpacing':36,'rankSpacing':44}}}%% flowchart LR - test["test
119 files"] - src["src
110 files"] - test["test
121 files"] - src["src
111 files"] + test["test
133 files"] + src["src
119 files"] landing["landing
61 files"] research["research
37 files"] + bench["bench
6 files"] global["global
5 files"] - bench["bench
3 files"] - scripts["scripts
2 files"] + scripts["scripts
3 files"] docs["docs
1 file"] examples["examples
1 file"] - test -- 247 --> src - test -- 244 --> src - bench -- 8 --> src + test -- 281 --> src + bench -- 12 --> src examples -- 4 --> src + test -- 3 --> global + test -- 3 --> scripts test -- 2 --> bench - test -- 2 --> global - test -- 2 --> scripts scripts --> src src --> global ``` diff --git a/CHANGELOG.md b/CHANGELOG.md index cf9300be..507dc9a4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,181 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +### Changed + +These tighten what a result is allowed to CLAIM, after the 2026-09-26 external deep review +reproduced 16 cases where a label (`PASS`, `complete`, `exact`, trusted) was stronger than the +evidence behind it. Scripts that read the JSON output may need to adapt: + +- **`forge verify` covers the whole repo, or says what it did not cover.** Every nested + package that declares its own suite (an explicit `scripts.test`, a pytest config, go.mod…) + is now planned and run in its own directory, and `tests.coverage` reports which packages + got a verdict. A passing root suite can no longer hide a failing workspace package (F08). + A root script that already runs every workspace (`npm test --workspaces`, `pnpm -r`, + `turbo run test`, …) covers them once, without duplicate runs. New `.forge/forge.config.json` + keys: `verify.workspaces: "root"` (declare that the root command covers everything), + `verify.exclude` (package paths that are not required suites) and `verify.generated` + (outputs a test run may legitimately write). Fixture/test-data packages are not required. +- **A test runner that is only a devDependency is inventory, not an obligation (F09).** With an + explicit `scripts.test`, `detectStack().testCommands` no longer adds `npx vitest`/`npx jest`; + they are listed in the new `testInventory` (and `forge stack` prints them as "available"). + npm's `"no test specified"` placeholder is not a suite. +- **`forge verify` refuses to bind a verdict to code that changed while the tests ran (F10).** + The code state is captured before and after the run; if it moved, the result is + `INCOMPLETE` with `mutated: true`, and the stamp is bound to the PRE-run state. Interpreter + caches (`__pycache__`, `.pytest_cache`, …) never count as a change. +- **The code-state fingerprint is a canonical manifest bound to HEAD (F01).** Renaming an + untracked file, moving bytes between files, adding an empty file, changing an exec bit or a + symlink target, and checking out another commit all change it; an unreadable untracked file + makes the state unbindable. Stamps from older forge versions no longer verify (by design: + their fingerprint could not tell those states apart) — re-run `forge verify`. +- **Exact reuse keys are lossless except whitespace (F04).** Case, operators, literals and + punctuation are part of the key, so `>= 18` / `<= 18`, `"ADMIN"` / `"admin"`, `= true` / + `!= true`, `getURL` / `getUrl` never share one. A near candidate must also pass a semantic + guard (same operators, numbers, literals, identifiers, paths and polarity words) or it is + only offered at the adapt tier. Artifacts minted before this change never exact-hit. +- **`forge context` reports delivery, not availability (F02, F03).** `tokens` is measured on the + rendered block (a chars/3.6 estimate, now labeled as such) and never exceeds `--budget` + while the result claims success: when even pointers cannot fit, items are dropped and the + result is `overflow: true`, `ok: false`. A `- read ` pointer is a `pending` read, not + coverage; a definition span / first-25-lines head covers only what it shows; omitted + dependents are listed in `truncated`. New `--block` prints the assembled context itself. +- **`forge route universal` makes an unreachable objective explicit (F12).** A budget no + cascade fits, or a target none reaches, returns `ok: false`, `feasible: false`, a `reason`, + `budgetMet: false` (budget objective) and `minimumExpectedCost`, with the least-bad cascade + only as a labeled `fallback`; the CLI prints `INFEASIBLE` and exits 1. Every recommendation + also reports `maxPossibleCost` (every attempt runs) next to the expected cost. +- **`forge route universal` says when a recommended model cannot be called (A07).** Each + cascade step lists the providers that serve it and where its cost comes from. A step that no + configured provider serves is marked `no provider id`, and the recommendation is labeled + `advice only` (`applicable: false`, `unmapped` in `--json`). Registry presence is not + availability: the shipped default recommendation was two such models. +- **`forge route outcome` validates what it records (A04, A01).** Unknown model ids, negative + or non-finite costs and wrong-length feature vectors are refused; each row carries an + `attemptId` (`--attempt ` makes re-recording idempotent) and is `self-reported` unless + `--verify-run ` ties it to a matching `forge verify` run. +- **`forge imagine --run` is described as what it is: an isolated checkout of HEAD, not a + security sandbox (A05)**, and a project whose tests use another runner (jest, vitest, + pytest…) gets an explicit unsupported-runner result instead of a `node --test` run of files + written for a different runner. + +- **Cost model: the 90.2/85.6/74.3% scenarios built on the refuted 0.62 routing factor are + withdrawn.** Every stage saving is labeled a hypothesis. A cost headline must now state its + run id, code SHA, dataset, denominator, baseline, correctness rule, uncertainty and evidence + status (`docs/plans/substrate-v2/05-cost-model.md` §3). +- **README leads with the three jobs**: shared evidence-referenced memory, heuristic + change-impact analysis, and explicit verification. P4 context assembly and P8 evaluation + are marked partial, and UI checks are documented as advisory. + +### Security + +- **The dashboard checks Host on every route, and writes need the page's session token and + this exact origin (F13).** A foreign Host (DNS rebinding) gets 403 on reads too; another + localhost port cannot write. Native clients that POSTed without a token must now send the + `x-forge-token` the server embeds in its page. Missing session logs render as "spend + unknown", never `$0`; unpriced models are "unpriced", not free (A10). + +### Fixed + +- **One evidence event, one vote (F06).** Four spellings of one commit (7/8/9/40 characters) + lifted a lesson from 0.655 to 0.821, across the 0.8 required-context bar. `val()` now counts + distinct events (an abbreviation of a git object id is the same event; a re-recorded ref is + the same event, and cannot refresh its decay), and `appendEvidence` stores resolvable + abbreviations under the full object id. +- **A rewritten lesson no longer inherits the old wording's trust (F07).** Evidence carries + over only when the rewrite is equivalent up to case, whitespace and punctuation; otherwise + the new claim starts at the 0.5 prior, names its parent (`supersedes`) and any semantic + conflicts in its provenance, and the legacy lesson restarts as a candidate. +- **Artifacts are revalidated where they are served (F05).** A cached artifact whose file was + edited or deleted is no longer served; dependency contracts recorded at mint (a + declaration fingerprint) invalidate an artifact when a same-name dependency's signature + changes; and "no atlas" is `unknown` validation (`requiresRevalidation: true`), never "ok". +- **Archived is not refuted (F15).** Pruning records WHY a claim went to the attic + (`tombstoned`, `dormant`, `idle`, `duplicate` + survivor, in `attic/.log`), and learned- + lesson consolidation drops a lesson only when its matching claim is actually retracted or + dormant — an idle archive keeps it, a deduplicated one defers to its survivor. +- **Similar-but-opposite rules are never merged (F16).** Consolidation and the ledger's + duplicate compaction only merge texts that pass the semantic guard; "Enable + authentication…" and "Disable authentication…" stay two claims and are reported as a + conflict (`forge ledger compact`, `learn-consolidate`). +- **Sparse router cost fits no longer invent a $1 intercept (F11).** One observed $0.05 attempt + used to predict ≈ $88.87; without enough data for slopes each model's intercept is its mean + log cost, with an explicit variance prior, and the fit reports `method`, `s2Source`, + `counts`, `alphaSE` and excluded zero-cost/missing rows. +- **The reuse benchmark measures the tiers it names (F14).** The fixture's evidence refs were + capped below the serving floor, so every "exact"/"near" row had timed a miss. Fixtures now + cite a real git object, each row is validated before timing (a stale fixture aborts the + run), cold rows clear every memoized sketch, and exact/near/miss and warm/cold are separate + rows. `reports/benchmarks.md` was regenerated; the landing page and README quote it. +- **`forge verify` no longer runs a project's suite with the parent test runner's + `NODE_TEST_CONTEXT`**, which made a nested `node --test` exit 0 without running its files. +- The format-on-edit guard no longer rewrites a Biome project with a global prettier. +- Evidence records with a non-finite day or an oversized ref, and malformed + `.forge/models.json` entries, are refused at the boundary (reported, never trusted). +- The dashboard reports an unreadable store as unreadable (`meta.errors`), not as empty. +- `forge substrate` says when the impact graph was truncated by the atlas file cap + (`capped`, `skippedFiles`), instead of presenting a partial graph as the whole repo (A06). +- `forge verify` no longer counts its own outputs under `.forge/` as changed files. +- `forge ledger verify --fix --dry-run` previews the address migration without writing. + +### Added + +- `bench/universal-router/reproduce.sh` rebuilds the shipped router prior from pinned, + sha256-checked public inputs (`sources.json`). The refit reproduces `data/router_prior.json` + exactly: all 176 fitted values, with only `fittedAt` different (Node v22.22.2, about 7 minutes + on 4 vCPUs). `holdout_eval.mjs` is a new seeded 150/350 held-out experiment. It is not a + reproduction of the reported 76.3% / $0.093 headline, whose harness is external. In it the + router did not beat a fixed cascade on solve rate (80.0% vs 80.6%) but was slightly cheaper. + It under-predicted cost by 5–22%, because failed attempts cost 1.2–2.0× as much as + successful ones. See `bench/universal-router/README.md`. +- `src/semantic_guard.js` (behaviour-bearing token comparison) and `src/schema.js` (narrow + runtime validation), both zero-dependency. +- Seeded property tests for the trust invariants (`test/trust_properties.test.js`) and a + regression test for each reproduced review finding. +- CI runs the two Python prototype suites (49 + 23 tests), the Theorem-D sanity checks and + the recomputation of every corrected research number from the shipped replication package. +- Command handlers for memory, verification and routing moved to `src/cli/` (a pure move; + `src/cli.js` keeps dispatch and help). + +### Documentation + +- A machine-readable claim/status registry (`docs/status/claims.json`, 47 claims assessed + against `d2abfa6`), with a generated table in `docs/status/README.md`. `node + scripts/claims-status.mjs --check`, now part of the CI quality gate, fails when the registry + is invalid, the table is stale, or a `docs/cognitive-substrate/` copy has drifted from its + `research/` source. +- `docs/INTEGRATIONS.md`: for every supported tool, config emission, MCP registration, + automatic hooks and enforcement are listed separately, each marked tested, declared or not + supported. It also records which registry models have provider ids, with a date. +- Universal router docs. The run-4 held-out headline is labeled repository-reported. The + shipped prior's refit is documented as reproduced exactly in this repository. The new + held-out replay is documented, as are the modeling limits: cascade cost under-predicted by + 5–22%, optimistic targets, and budgets that bound only expected cost. The dataset pin is + corrected to `SWE-bench/SWE-bench_Verified@78f471b`. +- Research corrections (2026-09-26) in the synthesis, the preprint and the white paper: + - Theorem D: joint attainability, with a counterexample. + - The equality condition of the silent-miss bound. + - "A caught miss is not a completed task." + - The frozen-model thesis restated to account for in-context learning. + - Prior art (CoALA, Reflexion). + - Impact-oracle results reported separately for each threshold. + + `research/recompute_corrections.py --theorem-checks` asserts the Theorem D checks without + data, and the recomputation also prints macro F1. +- Evidence grades are split into bibliographic verification, claim support, study design, + independent replication and transfer scope. METR's slowdown result is scoped to its + 16-developer, early-2025 study, with a link to the February 2026 update. +- The Qur'anic lens labels the Arabic source text, the translation, tafsir and the author's + design analogy separately, and states what the lens does and does not establish. No Arabic + text or translation was changed. +- The research PDFs are marked as historical, pre-correction editions and recorded in + `research/HISTORICAL_EDITIONS.md` (git blob, sha256, pinned commit, figure map, render + recipe). They were not re-rendered: the Qur'anic text in a fresh render could not be + verified, and the refutation paper needs a TeX toolchain. +- `forge verify`, `forge stack`, `forge ledger`, `forge reuse`, `forge dash` and `forge route + universal` sections of the guide describe the new coverage, binding, archive, reuse, + dashboard and advice-only behavior. + ## [1.4.3] - 2026-09-24 ### Fixed diff --git a/CITATION.cff b/CITATION.cff index 07081f27..0d65a8cd 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -4,12 +4,11 @@ title: "Forge (forgekit)" version: 1.4.3 date-released: "2026-09-24" abstract: >- - One brain for every AI coding agent — the cognitive substrate every frozen model is - missing. Forge gives a stateless model the memory, blast-radius foresight, and - guardrails it structurally lacks: proof-carrying team memory, an independent - verification gate, self-correcting project memory, a security-vetting skill-gate, and a - cost governor — authored once and emitted natively to Claude Code, Codex, Cursor, - Gemini, Aider, and more. + A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, + and explicit verification around coding agents: evidence-referenced ("proof-carrying") + team memory, a heuristic blast-radius analysis, a test-running verification gate, a + security-vetting skill-gate, and a cost meter — authored once and emitted natively to + Claude Code, Codex, Cursor, Gemini, Aider, and more. type: software authors: - name: CodeWithJuber diff --git a/ONBOARDING.md b/ONBOARDING.md index 1ffce1e7..e3b60335 100644 --- a/ONBOARDING.md +++ b/ONBOARDING.md @@ -1,11 +1,12 @@ # Onboarding — five minutes to productive -**One brain for every AI coding agent.** A language model is _stateless_ — one -context window, wiped every call — so it has no memory of what your team learned, no -foresight about what an edit breaks, and no enforced guardrails. forgekit is the -**cognitive substrate** that supplies exactly those three things, and it delivers them -as native config to Claude Code, Codex, Cursor, Gemini, Aider, Copilot, Windsurf, Zed, -Continue, and OpenClaw at once. Author the brain once; every tool reads it. +**A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, +and explicit verification around coding agents.** A language model keeps no durable state +between independent calls and sees only its context window, so on its own it does not carry +what your team learned, cannot see what an edit affects unless those files are in context, and +cannot enforce rules on itself. forgekit supplies that state and those checks from outside the +model, and delivers them as native config to Claude Code, Codex, Cursor, Gemini, Aider, Copilot, +Windsurf, Zed, Continue, and OpenClaw at once. Author it once; every tool reads it. This page is the fast path: install, configure a repo, do a task, and watch the ledger start paying off on day two. diff --git a/README.md b/README.md index 93591f82..b863c5ae 100644 --- a/README.md +++ b/README.md @@ -14,28 +14,39 @@

-Forge is one shared brain for your AI coding agents. It gives a stateless model the -three things it structurally lacks — memory, foresight, and guardrail hooks — and -delivers them into every tool you use. - -> An experimental reliability toolkit for AI-assisted coding — evidence-referenced, -> content-addressed memory (we call it "proof-carrying memory" / PCM — see the honesty note -> below), heuristic impact foresight, and guardrail hooks (automatic on Claude Code; -> instructions and MCP tools elsewhere) — authored once and delivered as native config to -> Claude Code, Codex, Cursor, Gemini, Aider, Copilot, Windsurf, Zed, Continue, and OpenClaw -> (plus MCP config for Roo and VS Code). Guardrails reduce risk; they are not a security -> sandbox. +**A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, and +explicit verification around coding agents.** Forgekit is a Node.js CLI and MCP server with no +runtime dependencies. You author project rules once, and it emits native configuration for Claude +Code, Codex, Cursor, Gemini, Aider, Copilot, Windsurf, Zed, Continue and OpenClaw (plus MCP +configuration for Roo and VS Code). Claude Code is the most deeply exercised integration. + +It does three jobs: + +1. **Share reliable project knowledge.** Lessons, facts and decisions live in the repository as + *evidence-referenced memory*: each claim carries references to its evidence, only tests, CI or + a person raise its confidence, and teammates merge it over plain git. +2. **Inspect likely change impact.** Before an edit, a *heuristic* impact analysis lists the files + the change probably touches, including coupled files you did not name. It reads a + regex-derived code graph, so it can miss files as well as over-warn. +3. **Verify the current change.** `forge verify` runs the repository's detected test suites and + reports `PASS`, `FAIL`, `INCOMPLETE` or `NOT_CONFIGURED` for the code state it ran on. A run that + covers only part of the repository is *partial* verification, and says so. > **Status: beta — read before you rely on it.** > -> - The core (`init`, `sync`, `substrate`, `impact`, `ledger`, guards) is tested and in daily -> use; some flags may change before `1.0`. +> - Releases follow Semantic Versioning ([CHANGELOG](CHANGELOG.md), +> [release notes](docs/RELEASING.md)): since 1.0.0 a breaking change ships as a new major +> version. "Beta" describes maturity, not interface churn — the analyses are heuristic, the +> checks are advisory unless you enable enforcement, and most integrations have less real-world +> exercise than Claude Code. The core (`init`, `sync`, `substrate`, `impact`, `ledger`, guards) +> is tested and in daily use. > - **Claude Code is the deepest-tested integration** (full plugin, ambient `UserPromptSubmit` -> guards). The other nine tools receive native config plus MCP tools, but have had less -> real-world exercise. On OpenClaw specifically, rules arrive via `AGENTS.md` project -> context; the config-only path uses a one-command MCP registration, while installing the -> package as a compatible Codex bundle loads its skills and bundle-scoped MCP server. -> Neither path provides ambient hooks (see +> guards). The other nine tools receive native config (most also an MCP server entry; Aider +> and Windsurf/Devin do not) but no automatic hooks, and have had less real-world exercise. +> On OpenClaw specifically, rules arrive via `AGENTS.md` project context; the config-only +> path uses a one-command MCP registration, while installing the package as a compatible +> Codex bundle loads its skills and bundle-scoped MCP server. Neither path provides ambient +> hooks (see > [OpenClaw in ARCHITECTURE](ARCHITECTURE.md#openclaw-what-is-automatic-and-what-is-not)). > - **Impact/blast-radius analysis is heuristic** — a regex-approximate code graph, not a sound > call graph. It is not conservative: it can miss affected files as well as flag unaffected @@ -49,49 +60,9 @@ delivers them into every tool you use. > not require `bash` on `PATH`: their Node launcher finds Git Bash and preserves guard exits. > `protect-paths` needs no bash at all and fails closed: when it cannot reach a verdict, the > tool call is blocked rather than let through. - -## Start in 60 seconds -Forgekit is a beta Node.js CLI and MCP server for AI-assisted software development. It -externalizes project memory, predicts the likely impact of code changes, and adds -deterministic checks around coding-agent workflows. The same source can emit native -configuration for several AI coding tools; Claude Code is the most deeply exercised -integration. - -The project is best read as **agent reliability and developer tooling**. It is not presented -as an enterprise multi-agent application, a general-purpose RAG platform, or an Azure AI -deployment. - -## Portfolio evidence - -For reviewers evaluating hands-on Agentic AI or GenAI work, each claim below links to the -implementation and its closest test or build proof. The evidence snapshot used for this -table is default-branch commit -[`3d9be37`](https://github.com/CodeWithJuber/forgekit/commit/3d9be37e26639c5c0a787d9196562b70444e2640). - -| Area | Implementation evidence | Test or delivery evidence | Evidence-safe claim | -| --- | --- | --- | --- | -| MCP tools | [`src/mcp_tools.js`](src/mcp_tools.js) defines 21 tool schemas; [`src/cortex_mcp.js`](src/cortex_mcp.js) implements JSON-RPC `initialize`, `tools/list`, and `tools/call` handlers | [`test/mcp.test.js`](test/mcp.test.js), [`test/cortex_mcp.test.js`](test/cortex_mcp.test.js), [current audited CI run](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393690) | Implemented an MCP server that exposes memory, preflight, routing, impact, verification, and health operations to compatible clients | -| Agent memory | [`src/ledger.js`](src/ledger.js) implements content-addressed claims, an oracle taxonomy, time-decayed validity, ranked retrieval, and a semilattice merge; [`src/ledger_store.js`](src/ledger_store.js) adds persistence, hash verification, and quarantine; [`src/ledger_sync.js`](src/ledger_sync.js) adds directory and git-ref sync | [`test/ledger.test.js`](test/ledger.test.js), [`test/ledger_store.test.js`](test/ledger_store.test.js), [`test/ledger_sync.test.js`](test/ledger_sync.test.js) | Implemented durable, evidence-weighted, mergeable memory for coding-agent workflows | -| LLM integration | [`src/llm.js`](src/llm.js) implements Anthropic Messages and OpenAI-compatible chat-completions calls; [`src/providers.js`](src/providers.js) configures Anthropic, OpenRouter, LiteLLM, OpenAI, Gemini, and custom endpoints | [`test/llm.test.js`](test/llm.test.js), [`test/providers.test.js`](test/providers.test.js) | Implemented direct, bounded single-prompt LLM adapters and provider configuration; this is not a streaming or autonomous tool-call client loop | -| Retrieval and embeddings | [`src/context.js`](src/context.js) assembles code definitions, dependants, tests, and trusted lessons under a token budget; [`src/embed.js`](src/embed.js) supports an optional command or OpenAI-compatible embedding endpoint, cosine similarity, and a disk cache; [`src/reuse.js`](src/reuse.js) falls back to MinHash and gates reuse on evidence | [`test/context.test.js`](test/context.test.js), [`test/embed.test.js`](test/embed.test.js), [`test/reuse.test.js`](test/reuse.test.js) | Implemented repository-local retrieval/context augmentation and an optional embedding adapter; no vector database or enterprise-document ingestion pipeline is claimed | -| Guardrails and verification | [`hooks/hooks.json`](hooks/hooks.json) wires lifecycle hooks; [`global/guards/protect-paths.sh`](global/guards/protect-paths.sh) and [`global/guards/secret-redact.sh`](global/guards/secret-redact.sh) add path and secret controls; [`src/skillgate.js`](src/skillgate.js), [`src/verify.js`](src/verify.js), and [`src/consensus.js`](src/consensus.js) implement scanning and multi-lens checks | [`test/secrets.test.js`](test/secrets.test.js), [`test/skillgate.test.js`](test/skillgate.test.js), [`test/verify.test.js`](test/verify.test.js), [`test/consensus.test.js`](test/consensus.test.js); [Security workflow](https://github.com/CodeWithJuber/forgekit/actions/workflows/security.yml) and [CodeQL](https://github.com/CodeWithJuber/forgekit/actions/workflows/codeql.yml) | Implemented deterministic defence-in-depth controls and evidence-producing verification; the regex guards are not a security sandbox | -| Human review affordances | [`src/ledger.js`](src/ledger.js) defines human accept/revert oracles; [`src/ledger_store.js`](src/ledger_store.js) implements ratify and retract records | [`test/ledger.test.js`](test/ledger.test.js), [`test/ledger_store.test.js`](test/ledger_store.test.js) | Implemented auditable human correction and ratification paths; no identity-enforced RBAC or enterprise approval workflow is claimed | -| Agent roles | [`.claude-plugin/plugin.json`](.claude-plugin/plugin.json) registers [`scout`](global/crew/scout.md), [`verifier`](global/crew/verifier.md), [`independent-reviewer`](global/crew/independent-reviewer.md), [`frontend-verifier`](global/crew/frontend-verifier.md), and [`doc-sync`](global/crew/doc-sync.md) | [`test/channels.test.js`](test/channels.test.js) checks plugin-channel wiring | Authored five concrete Claude Code role definitions; they are declarative roles, not a multi-agent orchestration runtime | -| Evaluation | [`src/eval.js`](src/eval.js) calculates precision, recall, and F1; [`bench/bench.mjs`](bench/bench.mjs) provides a seeded benchmark harness; [`reports/benchmarks.md`](reports/benchmarks.md) records methodology and limitations | [`test/eval.test.js`](test/eval.test.js), [`test/bench.test.js`](test/bench.test.js) | Implemented reproducible evaluation for the repository's impact predictor and local performance; the datasets are small and are not field benchmarks | -| Python research | [`research/python-prototypes/router_gate/`](research/python-prototypes/router_gate/) implements assumption gating, model routing, execution, verification, escalation, CLI, and MCP; [`research/python-prototypes/impact_oracle/`](research/python-prototypes/impact_oracle/) implements Python AST parsing and a persistent NetworkX dependency graph | [`router_gate` tests](research/python-prototypes/router_gate/tests/test_router_gate.py), [`router_gate` live demonstration results](research/python-prototypes/router_gate/eval_results.json), [`impact_oracle` tests](research/python-prototypes/impact_oracle/tests/test_demo_package.py) | Built working Python research prototypes; the shipped Forgekit runtime is Node and the Python packages are not presented as production services | -| Delivery engineering | [`package.json`](package.json) defines a Node 20+ CLI with no runtime dependencies; [`.github/workflows/release.yml`](.github/workflows/release.yml) gates releases and configures npm provenance; [`.github/workflows/smoke.yml`](.github/workflows/smoke.yml) exercises clean install and uninstall | [Release v0.32.1](https://github.com/CodeWithJuber/forgekit/releases/tag/v0.32.1); successful audited runs for [CI](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393690), [Smoke](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393550), [Security](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393540), [CodeQL](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393514), and [Scorecard](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393496) | Demonstrates packaging, cross-platform CI, security checks, and repeatable OSS release engineering; it does not establish enterprise production operation | - -### Maturity boundary - -| Evidence level | What belongs here | -| --- | --- | -| **Implemented and tested in the Node runtime** | CLI and config emitters; 21 MCP tools; agent memory and sync; code-context assembly; optional embedding adapter; LLM provider adapters; heuristic impact analysis; lifecycle guardrails; verification; benchmark harness; release automation | -| **Research or integration demonstration** | Five declarative Claude Code agent roles; Python router/gate and impact-oracle packages; a 30-task live routing demonstration; support for external embedding providers; configuration emitted for integrations other than the deeply tested Claude Code path | -| **Not claimed by this repository** | A collaborating multi-agent runtime; LangGraph, LangChain, Semantic Kernel, AutoGen, CrewAI, or Copilot Studio; Azure OpenAI or Azure AI Foundry; a vector database; enterprise-document RAG; business-system or RPA connectors; production Python deployment; multi-tenant cloud operation, SLA/SLO, Kubernetes, or infrastructure as code | - -Personal maintainer use is intentionally not used as proof of organizational adoption. No -customer count, enterprise deployment, production traffic, or service-level claim is made -without corresponding public evidence. +> - Guardrails reduce risk; they are not a security sandbox. What each integration actually does — +> config emission, MCP registration, automatic hooks, blocking — is in +> [docs/INTEGRATIONS.md](docs/INTEGRATIONS.md). ## 60-second quickstart @@ -108,26 +79,35 @@ forge substrate "Change verifyToken in src/auth.js to require length > 20; updat ``` The result includes an assumption verdict, a model-tier recommendation, predicted impact, -context completeness, scope clusters, and a verification checklist. On Claude Code, the +a context-coverage check (what was delivered, not a proof that it suffices), scope clusters, and +a verification checklist. On Claude Code, the pre-action check can run automatically through a `UserPromptSubmit` hook. On other supported tools, Forgekit emits instructions and MCP configuration that the tool can invoke. +The three jobs, one command each: + +```bash +forge ledger query "token validation" # 1. what does the team already know about this area? +forge impact verifyToken # 2. which files will this edit probably touch? +forge verify # 3. did the change pass the suites that actually ran? +``` + ## Contents -- [Portfolio evidence](#portfolio-evidence) -- [Maturity boundary](#maturity-boundary) - [60-second quickstart](#60-second-quickstart) - [Why Forgekit exists](#why-forgekit-exists) - [How the loop works](#how-the-loop-works) - [Core capabilities](#core-capabilities) -- [LLMs, retrieval, and embeddings](#llms-retrieval-and-embeddings) -- [Agent roles and MCP tools](#agent-roles-and-mcp-tools) -- [Measured evidence](#measured-evidence) - [Setup details](#setup-details) - [Commands](#commands) - [Team memory](#team-memory) -- [Structural comparison](#structural-comparison) +- [LLMs, retrieval, and embeddings](#llms-retrieval-and-embeddings) +- [Agent roles and MCP tools](#agent-roles-and-mcp-tools) - [Honest limits](#honest-limits) +- [Portfolio evidence](#portfolio-evidence) +- [Maturity boundary](#maturity-boundary) +- [Measured evidence](#measured-evidence) +- [Structural comparison](#structural-comparison) - [Python research prototypes](#python-research-prototypes) - [White paper](#white-paper) - [Public site](#public-site) @@ -170,14 +150,14 @@ so a wrong lesson decays out instead of ossifying. Full design: ## What you get -The day-to-day value first — the substrate gives a frozen model what it can't hold itself: +The day-to-day value first — what the substrate keeps outside the model, so it survives between sessions and tools: - **Memory that persists across sessions and teammates.** _[Implemented]_ Every lesson, fact, and verified reuse is _proof-carrying memory (PCM)_ — our name for **evidence-referenced, content-addressed memory**: a claim that carries references to its own evidence and is only trusted once independent oracles raise its confidence above a floor (the "proof" is that evidence trail, not a formal proof). Wrong lessons decay out instead of ossifying. -- **Foresight before you break things.** _[Heuristic]_ Ask "what does changing `verifyToken` +- **Heuristic impact before you break things.** _[Heuristic]_ Ask "what does changing `verifyToken` break?" and get the _blast radius_ — the set of files an edit is predicted to impact, read from a regex-approximate code graph (not sound, and it can miss affected files), including coupled files you never named. @@ -204,14 +184,14 @@ Every number is a median from `npm run bench` on this repo, recorded with its en block in [`reports/benchmarks.md`](reports/benchmarks.md) — the project rule is _a number is an assumption until measured_. -- **Blast radius in 0.40 ms** (warm code-graph). On 6 hand-labeled cases from this repo's - real import graph, recall is 1.00 against 0.27 for looking at the edited file alone, and +- **Blast radius in 1.68 ms** (warm code-graph). On 6 hand-labeled cases from this repo's + real import graph, recall is 1.00 against 0.26 for looking at the edited file alone, and precision is 0.17 — `impact` walks reverse dependencies transitively by default, so it returns everything downstream while the labels name only the direct referencers (restricted to one hop the same cases return their labeled sets). The precision 0.90 this line used to quote does not reproduce. On nine real Python repositories the research prototype's impact oracle reached recall 0.022 ([refutation](research/empirical-refutation/)). -- **A full pre-action gate in 886 ms** (median on this repo, warm, on a 4-core Windows VM — this +- **A full pre-action gate in 851 ms** (median on this repo, warm, on a 4-core Linux VM — this row is machine-bound; see the environment block) — assumption check, routing, reuse lookup, context assembly, blast radius, scope, and goal anchor in one deterministic pass, no LLM call. On Claude Code it runs on **every prompt, automatically**. @@ -221,8 +201,8 @@ an assumption until measured_. judge accepted it cost $1.06 against $1.76, but only 6 and 3 of 64 outputs were accepted ([refutation](research/empirical-refutation/)). `forge cost --stages` reports only _your_ measured stages. -- **Conflict-free team memory** — merging two 500-claim ledger replicas takes **4308 ms** on that - same VM (I/O-bound, 4–6x a Linux host); the +- **Conflict-free team memory** — merging two 500-claim ledger replicas takes **191 ms** on that + same VM (I/O-bound, so it varies with the disk); the merge is order-independent and property-tested, so teammate ledgers converge to the same state no matter who syncs first, over plain git. The substrate is advisory by default. Set `FORGE_ENFORCE=1` to block only its strongest @@ -245,126 +225,30 @@ from a fresh repository graph. candidates. - **Budgeted context assembly.** Definitions, direct dependants, sibling tests, and trusted lessons are selected under a token budget. Missing required context becomes a question - rather than invented context. + rather than invented context. Coverage is syntactic — delivered, not proven sufficient — token + counts are estimates, and a file that only fits as a "read this" pointer stays a pending read. - **Model-tier recommendation.** A deterministic rubric combines task text and repository signals. An optional LLM proposal can only lower the tier, confidence-gated and bounded; a vote for a higher tier is never applied automatically — it is recorded as an advisory `escalateTo` recommendation, which names the tier only once an external check has actually failed (the doom-loop diagnosis at its thrash threshold). Forgekit advises which tier to - request; it does not itself proxy or fail over model traffic. + request; it does not itself proxy or fail over model traffic. The opt-in cross-provider router + (`route universal`) minimises **expected routing cost** for a success target; its budgets are + expectations, not spend caps, and its benchmark headline is repository-reported. Its shipped + prior refits exactly from pinned public data in this repository, and a new in-repo held-out + replay finds it no better than a fixed cascade chosen on the same dev tasks + ([docs/UNIVERSAL_ROUTING.md](docs/UNIVERSAL_ROUTING.md)). - **Proof-gated reuse.** Cached code is served only after evidence clears a confidence floor and declared dependencies still resolve in the current repository graph. - **Lifecycle guardrails.** Claude Code hooks cover prompt preflight, protected paths, cost budget, repeated failures, format-on-edit, secret redaction, completion checks, and session learning. These controls reduce risk; they do not create a secure execution boundary. -- **Independent verification.** `forge verify` runs the repository's detected test suites, - reports `PASS`, `FAIL`, `INCOMPLETE`, or `NOT_CONFIGURED` honestly, checks unknown symbols, - and binds provenance to the code state. `--deep` adds structural, security, spec-drift, +- **Verification, partial or full.** `forge verify` runs the repository's detected test + suites, reports `PASS`, `FAIL`, `INCOMPLETE`, or `NOT_CONFIGURED`, checks unknown symbols, + and binds provenance to the code state it ran on. A verdict covers only the suites that ran: + partial coverage is `INCOMPLETE`, not `PASS`. `--deep` adds structural, security, spec-drift, impact, and optional model-review lenses. -## LLMs, retrieval, and embeddings - -### Direct LLM adapters - -`src/llm.js` supports two wire formats: - -- Anthropic Messages API; -- OpenAI-compatible chat completions, used for OpenAI, Gemini, OpenRouter, and LiteLLM. - -Credentials are read from environment variables and are not put in command-line arguments. -The implementation performs a bounded single-prompt request. It does not implement streaming, -conversation persistence, or a model-driven tool-call execution loop. - -### Repository-local retrieval - -Forgekit retrieves from its code graph and evidence ledger, then assembles a context bundle -for an external coding agent. This is retrieval and augmentation for source-code work, not a -general enterprise RAG pipeline. There is no document connector layer, chunking service, -citation generator, managed index, or vector database in this repository. - -### Optional embeddings - -MinHash remains the zero-dependency default. To opt into semantic similarity, configure an -external provider with one of the formats the current implementation accepts: - -```bash -# OpenAI-compatible embedding endpoint -export FORGE_EMBED="https://api.example.com/v1/embeddings" -export FORGE_EMBED_MODEL="your-embedding-model" -export FORGE_EMBED_KEY="your-provider-key" - -# Or a local/external command that implements Forgekit's stdin/stdout vector protocol -export FORGE_EMBED="cmd:./my-embedding-provider" -``` - -Vectors are cached in `.forge/embed-cache.jsonl`. Provider failure, timeout, malformed output, -or missing vectors falls back to MinHash. The repository test uses a deterministic fake -provider; it verifies adapter and fallback behavior, not live performance of a hosted model. - -The performance snapshot in [`reports/benchmarks.md`](reports/benchmarks.md) was generated at -commit `eb68ea9` and does not measure the optional embedding path. Current -embedding behavior is defined by [`src/embed.js`](src/embed.js) and -[`test/embed.test.js`](test/embed.test.js). - -## Agent roles and MCP tools - -Forgekit registers five Claude Code role definitions: - -- `scout` — read-only repository investigation; -- `verifier` — fresh-context correctness review; -- `independent-reviewer` — diff/spec/test-only merge gate; -- `frontend-verifier` — visual and accessibility review; -- `doc-sync` — documentation consistency after code changes. - -These are concrete role prompts supplied to the host tool. Forgekit does not contain a runtime -that spawns these roles, exchanges inter-agent messages, or executes a LangGraph-style state -machine. - -The built-in MCP server exposes 21 tools, including: - -- pre-action checks: `substrate_check`, `preflight_check`, `assumption_gate`, `route_task`; -- repository analysis: `predict_impact`, `scope_files`, `rank_code`, `collide_check`; -- memory: `cortex_lessons`, `forge_remember`, `forge_ledger_query`, - `forge_ledger_ratify`, `forge_ledger_retract`; -- operations: `forge_doctor`, `forge_provider_status`, `forge_cost`, dashboard data. - -The MCP server is the tool-provider side of function calling. The compatible host remains -responsible for deciding when to call a tool and feeding the result back to its model. - -## Measured evidence - -Numbers below are reported only with their test boundary. See -[`reports/benchmarks.md`](reports/benchmarks.md) and the linked evaluation artifacts for full -methodology. - -Parser-stable snapshot labels used by the generated project pages are: - -- **A full pre-action gate in 886 ms median** — deterministic, warm repository graph, LLM disabled; -- **Blast radius in 0.40 ms median** — warm impact query; and -- **20.2% more cost than always-premium** — the held-out routing result. The 62.1% saving the - white paper reported came from a 30-task demonstration with thresholds tuned on those same - tasks; on 80 pre-registered held-out tasks the same router spent 20.2% *more* (table below). - Per judged-correct output the pipeline cost $1.06 against always-premium's $1.76. - -The boundaries in the table below are part of each result. - -| Measurement | Recorded result | Boundary | -| --- | ---: | --- | -| Warm impact query | 0.40 ms median | 30 runs on one JavaScript repository with a memoized adjacency index; not model latency | -| Deterministic substrate check | 886 ms median | 3 runs on one repository, warm graph, LLM disabled, on a 4-core Windows VM — wall-clock rows are machine-bound and were ~150 ms on the Linux host that produced the pre-2026-09-22 snapshot; re-run `npm run bench` on your own hardware | -| Impact quality | precision 0.17, recall 1.00, F1 0.29 (the precision 0.90 / F1 0.92 reported before 2026-09-21 do not reproduce) | 6 hand-labelled symbols in this repository (which imports only by relative path, so it does not exercise tsconfig path aliases), scored by `evalImpact` against labels re-derived by `git grep`; `impact` walks reverse dependencies transitively by default, so precision measures the transitive closure against direct-only labels; edited-file-only baseline recall 0.27 | -| Ledger replica merge | 4308 ms median | 3 runs merging two synthetic 500-claim replicas with 250 claims shared, on the same 4-core Windows VM (I/O-bound: 4–6x the Linux host's figure) | -| Python router live demonstration | 62.1% calculated cost reduction versus always-premium | 30 hand-labelled tasks, thresholds tuned to the set, real measured LLM tokens, approximate public prices; demonstration, not field benchmark | -| Python router, held-out evaluation | total spend 20.2% **higher** than always-premium; gate F1 0.37 | 80 tasks from real GitHub issues and PRs, thresholds frozen, pre-registered; refutes the row above | -| Python impact oracle | precision 0.633, recall 1.000, F1 0.753 | 5 mutations in the bundled demo package; mutation-derived test failures as ground truth | -| Python impact oracle, real repositories | precision 0.398, recall 0.022, F1 0.042 (grep baseline F1 0.437) | 759 files in 9 open-source repositories, co-change ground truth, pre-registered; refutes the row above | - -The current audited CI run at commit `3d9be37` completed successfully for Node 20, Node 22, -and Windows Git Bash, plus the reusable quality gate. The quality gate ran the Node unit suite, -Biome checks, TypeScript type checking, critical-level npm audit, ShellCheck, zero-runtime-dependency -assertion, version and documentation checks, and `npm pack --dry-run`. The Python prototype -pytest suites are present in the repository but are not part of that current CI workflow. - ## Setup details Install using one path: @@ -426,7 +310,7 @@ and output live in [`docs/GUIDE.md`](docs/GUIDE.md). | | `forge preflight` | assumption check — what a task names that the repo doesn't define | | | `forge impact` | hazard-aware blast radius — SCC-aware propagation + data-driven threshold from PageRank centrality and ledger incident history | | | `forge scope` | decompose files into independent clusters (+ coupled files you didn't name) | -| | `forge context` | budgeted context assembly + completeness gate — what an edit NEEDS known | +| | `forge context` | budgeted context assembly + completeness gate — what an edit NEEDS known, delivered or owed | | | `forge route` | recommend the cheapest capable model for a task (+ gateway config); `route universal`: any provider's models, lowest expected cost for the success asked for, learned from outcomes | | | `forge verify` | independent verification gate — tests + hallucinated-symbol + provenance (--deep: multi-lens consensus) | | | `forge precommit` | commit-level gate — staged code w/o docs + secret scan (FORGE_COMMIT_GATE=block|warn|0) | @@ -488,32 +372,93 @@ repository's git remote. A non-fast-forward race triggers a re-merge and bounded directory can be selected with `--dir ` or `FORGE_SYNC_DIR`; `--personal` includes the per-user ledger. -## Structural comparison +## LLMs, retrieval, and embeddings -This table describes architecture, not a claim of superiority or equivalent product scope. +### Direct LLM adapters -| Concern | Forgekit implements | Boundary | -| --- | --- | --- | -| Memory | Content-addressed claims, provenance, oracle-weighted validity, time decay, and git-native union merge | No hosted synchronization, managed database, enterprise tenancy, or RBAC | -| Retrieval | MinHash retrieval by default; optional external embeddings; proof-gated code reuse | No vector database or general enterprise corpus pipeline | -| Routing | A visible deterministic recommendation with optional bounded LLM input; LiteLLM alias config emission | Does not proxy traffic, manage quotas, perform failover, or hold provider keys | -| Tool use | A JSON-RPC MCP server with 21 executable tool handlers | Does not implement the model/client loop that chooses and executes tool calls autonomously | -| Agent roles | Five host-consumable Claude Code role definitions | No multi-agent graph, scheduler, inter-agent messaging layer, or named orchestration framework | -| Verification | Repository tests, provenance, structural and security lenses, limited benchmark suites | No managed GenAI evaluation service, production telemetry, online evaluation, or broad red-team certification | +`src/llm.js` supports two wire formats: + +- Anthropic Messages API; +- OpenAI-compatible chat completions, used for OpenAI, Gemini, OpenRouter, and LiteLLM. + +Credentials are read from environment variables and are not put in command-line arguments. +The implementation performs a bounded single-prompt request. It does not implement streaming, +conversation persistence, or a model-driven tool-call execution loop. + +### Repository-local retrieval + +Forgekit retrieves from its code graph and evidence ledger, then assembles a context bundle +for an external coding agent. This is retrieval and augmentation for source-code work, not a +general enterprise RAG pipeline. There is no document connector layer, chunking service, +citation generator, managed index, or vector database in this repository. + +### Optional embeddings + +MinHash remains the zero-dependency default. To opt into semantic similarity, configure an +external provider with one of the formats the current implementation accepts: + +```bash +# OpenAI-compatible embedding endpoint +export FORGE_EMBED="https://api.example.com/v1/embeddings" +export FORGE_EMBED_MODEL="your-embedding-model" +export FORGE_EMBED_KEY="your-provider-key" + +# Or a local/external command that implements Forgekit's stdin/stdout vector protocol +export FORGE_EMBED="cmd:./my-embedding-provider" +``` + +Vectors are cached in `.forge/embed-cache.jsonl`. Provider failure, timeout, malformed output, +or missing vectors falls back to MinHash. The repository test uses a deterministic fake +provider; it verifies adapter and fallback behavior, not live performance of a hosted model. + +The performance snapshot in [`reports/benchmarks.md`](reports/benchmarks.md) was generated at +commit `eb68ea9` and does not measure the optional embedding path. Current +embedding behavior is defined by [`src/embed.js`](src/embed.js) and +[`test/embed.test.js`](test/embed.test.js). + +## Agent roles and MCP tools + +Forgekit registers five Claude Code role definitions: + +- `scout` — read-only repository investigation; +- `verifier` — fresh-context correctness review; +- `independent-reviewer` — diff/spec/test-only merge gate; +- `frontend-verifier` — visual and accessibility review; +- `doc-sync` — documentation consistency after code changes. + +These are concrete role prompts supplied to the host tool. Forgekit does not contain a runtime +that spawns these roles, exchanges inter-agent messages, or executes a LangGraph-style state +machine. + +The built-in MCP server exposes 21 tools, including: + +- pre-action checks: `substrate_check`, `preflight_check`, `assumption_gate`, `route_task`; +- repository analysis: `predict_impact`, `scope_files`, `rank_code`, `collide_check`; +- memory: `cortex_lessons`, `forge_remember`, `forge_ledger_query`, + `forge_ledger_ratify`, `forge_ledger_retract`; +- operations: `forge_doctor`, `forge_provider_status`, `forge_cost`, dashboard data. + +The MCP server is the tool-provider side of function calling. The compatible host remains +responsible for deciding when to call a tool and feeding the result back to its model. ## Honest limits -- **Beta software.** The CLI is released and CI-tested, but interfaces may still change before - `1.0`. Support is maintainer-led and best-effort; there is no SLA. +- **Beta software.** The CLI is released, CI-tested and semantically versioned (a breaking + change is a major version); "beta" is about maturity and evidence, not about interface churn. + Support is maintainer-led and best-effort; there is no SLA. - **Claude Code is the deepest-tested integration.** Other emitters and MCP configuration are implemented, but the repository does not provide equivalent real-world exercise evidence for every supported host. - **The code graph is heuristic.** Regex extraction is not a sound call graph and can both over-predict and miss dependencies. - **Routing is advisory.** Forgekit recommends a tier. LiteLLM or another gateway must be run - separately to move traffic, handle failover, enforce quotas, or manage credentials. + separately to move traffic, handle failover, enforce quotas, or manage credentials. Routing + costs are expected values, not caps, and no cost saving has been measured end to end: the + ~90 % figure in the plans is a target (a hypothesis), and the white paper's 62.1 % routing + saving was refuted on held-out tasks. - **Memory is external state, not model training.** Claims live in files and are retrieved into - context; Forgekit does not update model weights. + context; Forgekit does not update model weights. It supplies persistence and external checks + that a model does not guarantee on its own — one tested way to do that, not the only one. - **Embeddings are an adapter, not an included model or vector store.** A user supplies the external command or HTTP service. MinHash remains the default and fallback. - **Guardrails are defence in depth.** Shell-regex checks can be bypassed and post-tool redaction @@ -527,6 +472,92 @@ This table describes architecture, not a claim of superiority or equivalent prod business-system/RPA connector layer, multi-tenant service, Kubernetes deployment, IaC stack, or public production-usage case study in this repository. +## Portfolio evidence + +The project is best read as **agent reliability and developer tooling**. It is not presented +as an enterprise multi-agent application, a general-purpose RAG platform, or an Azure AI +deployment. Per-claim status (implemented, measured, refuted, hypothesis, partial, reported) is +kept in [docs/status/](docs/status/README.md). + +For reviewers evaluating hands-on Agentic AI or GenAI work, each claim below links to the +implementation and its closest test or build proof. The evidence snapshot used for this +table is default-branch commit +[`3d9be37`](https://github.com/CodeWithJuber/forgekit/commit/3d9be37e26639c5c0a787d9196562b70444e2640). + +| Area | Implementation evidence | Test or delivery evidence | Evidence-safe claim | +| --- | --- | --- | --- | +| MCP tools | [`src/mcp_tools.js`](src/mcp_tools.js) defines 21 tool schemas; [`src/cortex_mcp.js`](src/cortex_mcp.js) implements JSON-RPC `initialize`, `tools/list`, and `tools/call` handlers | [`test/mcp.test.js`](test/mcp.test.js), [`test/cortex_mcp.test.js`](test/cortex_mcp.test.js), [current audited CI run](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393690) | Implemented an MCP server that exposes memory, preflight, routing, impact, verification, and health operations to compatible clients | +| Agent memory | [`src/ledger.js`](src/ledger.js) implements content-addressed claims, an oracle taxonomy, time-decayed validity, ranked retrieval, and a semilattice merge; [`src/ledger_store.js`](src/ledger_store.js) adds persistence, hash verification, and quarantine; [`src/ledger_sync.js`](src/ledger_sync.js) adds directory and git-ref sync | [`test/ledger.test.js`](test/ledger.test.js), [`test/ledger_store.test.js`](test/ledger_store.test.js), [`test/ledger_sync.test.js`](test/ledger_sync.test.js) | Implemented durable, evidence-weighted, mergeable memory for coding-agent workflows | +| LLM integration | [`src/llm.js`](src/llm.js) implements Anthropic Messages and OpenAI-compatible chat-completions calls; [`src/providers.js`](src/providers.js) configures Anthropic, OpenRouter, LiteLLM, OpenAI, Gemini, and custom endpoints | [`test/llm.test.js`](test/llm.test.js), [`test/providers.test.js`](test/providers.test.js) | Implemented direct, bounded single-prompt LLM adapters and provider configuration; this is not a streaming or autonomous tool-call client loop | +| Retrieval and embeddings | [`src/context.js`](src/context.js) assembles code definitions, dependants, tests, and trusted lessons under a token budget; [`src/embed.js`](src/embed.js) supports an optional command or OpenAI-compatible embedding endpoint, cosine similarity, and a disk cache; [`src/reuse.js`](src/reuse.js) falls back to MinHash and gates reuse on evidence | [`test/context.test.js`](test/context.test.js), [`test/embed.test.js`](test/embed.test.js), [`test/reuse.test.js`](test/reuse.test.js) | Implemented repository-local retrieval/context augmentation and an optional embedding adapter; no vector database or enterprise-document ingestion pipeline is claimed | +| Guardrails and verification | [`hooks/hooks.json`](hooks/hooks.json) wires lifecycle hooks; [`global/guards/protect-paths.sh`](global/guards/protect-paths.sh) and [`global/guards/secret-redact.sh`](global/guards/secret-redact.sh) add path and secret controls; [`src/skillgate.js`](src/skillgate.js), [`src/verify.js`](src/verify.js), and [`src/consensus.js`](src/consensus.js) implement scanning and multi-lens checks | [`test/secrets.test.js`](test/secrets.test.js), [`test/skillgate.test.js`](test/skillgate.test.js), [`test/verify.test.js`](test/verify.test.js), [`test/consensus.test.js`](test/consensus.test.js); [Security workflow](https://github.com/CodeWithJuber/forgekit/actions/workflows/security.yml) and [CodeQL](https://github.com/CodeWithJuber/forgekit/actions/workflows/codeql.yml) | Implemented deterministic defence-in-depth controls and evidence-producing verification; the regex guards are not a security sandbox | +| Human review affordances | [`src/ledger.js`](src/ledger.js) defines human accept/revert oracles; [`src/ledger_store.js`](src/ledger_store.js) implements ratify and retract records | [`test/ledger.test.js`](test/ledger.test.js), [`test/ledger_store.test.js`](test/ledger_store.test.js) | Implemented auditable human correction and ratification paths; no identity-enforced RBAC or enterprise approval workflow is claimed | +| Agent roles | [`.claude-plugin/plugin.json`](.claude-plugin/plugin.json) registers [`scout`](global/crew/scout.md), [`verifier`](global/crew/verifier.md), [`independent-reviewer`](global/crew/independent-reviewer.md), [`frontend-verifier`](global/crew/frontend-verifier.md), and [`doc-sync`](global/crew/doc-sync.md) | [`test/channels.test.js`](test/channels.test.js) checks plugin-channel wiring | Authored five concrete Claude Code role definitions; they are declarative roles, not a multi-agent orchestration runtime | +| Evaluation | [`src/eval.js`](src/eval.js) calculates precision, recall, and F1; [`bench/bench.mjs`](bench/bench.mjs) provides a seeded benchmark harness; [`reports/benchmarks.md`](reports/benchmarks.md) records methodology and limitations | [`test/eval.test.js`](test/eval.test.js), [`test/bench.test.js`](test/bench.test.js) | Implemented reproducible evaluation for the repository's impact predictor and local performance; the datasets are small and are not field benchmarks | +| Python research | [`research/python-prototypes/router_gate/`](research/python-prototypes/router_gate/) implements assumption gating, model routing, execution, verification, escalation, CLI, and MCP; [`research/python-prototypes/impact_oracle/`](research/python-prototypes/impact_oracle/) implements Python AST parsing and a persistent NetworkX dependency graph | [`router_gate` tests](research/python-prototypes/router_gate/tests/test_router_gate.py), [`router_gate` live demonstration results](research/python-prototypes/router_gate/eval_results.json), [`impact_oracle` tests](research/python-prototypes/impact_oracle/tests/test_demo_package.py) | Built working Python research prototypes; the shipped Forgekit runtime is Node and the Python packages are not presented as production services | +| Delivery engineering | [`package.json`](package.json) defines a Node 20+ CLI with no runtime dependencies; [`.github/workflows/release.yml`](.github/workflows/release.yml) gates releases and configures npm provenance; [`.github/workflows/smoke.yml`](.github/workflows/smoke.yml) exercises clean install and uninstall | [Release v0.32.1](https://github.com/CodeWithJuber/forgekit/releases/tag/v0.32.1); successful audited runs for [CI](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393690), [Smoke](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393550), [Security](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393540), [CodeQL](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393514), and [Scorecard](https://github.com/CodeWithJuber/forgekit/actions/runs/33693393496) | Demonstrates packaging, cross-platform CI, security checks, and repeatable OSS release engineering; it does not establish enterprise production operation | + +### Maturity boundary + +| Evidence level | What belongs here | +| --- | --- | +| **Implemented and tested in the Node runtime** | CLI and config emitters; 21 MCP tools; agent memory and sync; code-context assembly; optional embedding adapter; LLM provider adapters; heuristic impact analysis; lifecycle guardrails; verification; benchmark harness; release automation | +| **Research or integration demonstration** | Five declarative Claude Code agent roles; Python router/gate and impact-oracle packages; a 30-task live routing demonstration; support for external embedding providers; configuration emitted for integrations other than the deeply tested Claude Code path | +| **Not claimed by this repository** | A collaborating multi-agent runtime; LangGraph, LangChain, Semantic Kernel, AutoGen, CrewAI, or Copilot Studio; Azure OpenAI or Azure AI Foundry; a vector database; enterprise-document RAG; business-system or RPA connectors; production Python deployment; multi-tenant cloud operation, SLA/SLO, Kubernetes, or infrastructure as code | + +Personal maintainer use is intentionally not used as proof of organizational adoption. No +customer count, enterprise deployment, production traffic, or service-level claim is made +without corresponding public evidence. + +## Measured evidence + +Numbers below are reported only with their test boundary. See +[`reports/benchmarks.md`](reports/benchmarks.md) and the linked evaluation artifacts for full +methodology. + +Parser-stable snapshot labels used by the generated project pages are: + +- **A full pre-action gate in 851 ms median** — deterministic, warm repository graph, LLM disabled; +- **Blast radius in 1.68 ms median** — warm impact query; and +- **20.2% more cost than always-premium** — the held-out routing result. The 62.1% saving the + white paper reported came from a 30-task demonstration with thresholds tuned on those same + tasks; on 80 pre-registered held-out tasks the same router spent 20.2% *more* (table below). + Per judged-correct output the pipeline cost $1.06 against always-premium's $1.76. + +The boundaries in the table below are part of each result. + +| Measurement | Recorded result | Boundary | +| --- | ---: | --- | +| Warm impact query | 1.68 ms median | 30 runs on one JavaScript repository with a memoized adjacency index; not model latency | +| Deterministic substrate check | 851 ms median | 3 runs on one repository, warm graph, LLM disabled, on a 4-core Linux VM (Node v22.22.2, recorded 2026-09-26 in the environment block) — wall-clock rows are machine-bound; re-run `npm run bench` on your own hardware | +| Impact quality | precision 0.17, recall 1.00, F1 0.28 (the precision 0.90 / F1 0.92 reported before 2026-09-21 do not reproduce) | 6 hand-labelled symbols in this repository (which imports only by relative path, so it does not exercise tsconfig path aliases), scored by `evalImpact` against labels re-derived by `git grep`; `impact` walks reverse dependencies transitively by default, so precision measures the transitive closure against direct-only labels; edited-file-only baseline recall 0.26 | +| Ledger replica merge | 191 ms median | 3 runs merging two synthetic 500-claim replicas with 250 claims shared, on the same 4-core Linux VM (I/O-bound: varies with the disk more than the CPU-bound rows do) | +| Python router live demonstration | 62.1% calculated cost reduction versus always-premium (repriced tokens; refuted by the next row) | 30 hand-labelled tasks, thresholds tuned to the set, real measured LLM tokens, approximate public prices; demonstration, not field benchmark | +| Python router, held-out evaluation | total spend 20.2% **higher** than always-premium; gate F1 0.37; success = judge-accepted (6/64 vs 3/64), not test-verified | 80 tasks from real GitHub issues and PRs, thresholds frozen, pre-registered; refutes the row above | +| Python impact oracle | precision 0.633, recall 1.000, F1 0.753 | 5 mutations in the bundled demo package; mutation-derived test failures as ground truth | +| Python impact oracle, real repositories | precision 0.398, recall 0.022, F1 0.042 (grep baseline F1 0.437) | 759 files in 9 open-source repositories, co-change ground truth, pre-registered; refutes the row above | + +The current audited CI run at commit `3d9be37` completed successfully for Node 20, Node 22, +and Windows Git Bash, plus the reusable quality gate. The quality gate ran the Node unit suite, +Biome checks, TypeScript type checking, critical-level npm audit, ShellCheck, zero-runtime-dependency +assertion, version and documentation checks, and `npm pack --dry-run`. The Python prototype +pytest suites were not part of CI at that commit; since `aedddf5` a separate CI job runs both +prototype suites, the Theorem D sanity checks (`research/recompute_corrections.py +--theorem-checks`) and the recomputation of every corrected number from the replication package. + +## Structural comparison + +This table describes architecture, not a claim of superiority or equivalent product scope. + +| Concern | Forgekit implements | Boundary | +| --- | --- | --- | +| Memory | Content-addressed claims, provenance, oracle-weighted validity, time decay, and git-native union merge | No hosted synchronization, managed database, enterprise tenancy, or RBAC | +| Retrieval | MinHash retrieval by default; optional external embeddings; proof-gated code reuse | No vector database or general enterprise corpus pipeline | +| Routing | A visible deterministic recommendation with optional bounded LLM input; LiteLLM alias config emission | Does not proxy traffic, manage quotas, perform failover, or hold provider keys | +| Tool use | A JSON-RPC MCP server with 21 executable tool handlers | Does not implement the model/client loop that chooses and executes tool calls autonomously | +| Agent roles | Five host-consumable Claude Code role definitions | No multi-agent graph, scheduler, inter-agent messaging layer, or named orchestration framework | +| Verification | Repository tests, provenance, structural and security lenses, limited benchmark suites | No managed GenAI evaluation service, production telemetry, online evaluation, or broad red-team certification | + ## Python research prototypes The Python packages under [`research/python-prototypes/`](research/python-prototypes/) are the @@ -553,8 +584,13 @@ that set, not a general accuracy estimate. Despite legacy package metadata using ### Impact Oracle [`impact_oracle`](research/python-prototypes/impact_oracle/) parses Python ASTs, stores a NetworkX -dependency graph as JSON, and predicts change impact through reverse-dependency traversal. The -bundled evaluation mutates five symbols and uses pytest failures as independent ground truth. +dependency graph as JSON, and predicts change impact through reverse-dependency traversal plus the +repaired sibling and forward relations (the in-tree package is the repaired v2, 49 tests). The +bundled evaluation mutates five symbols and uses pytest failures as independent ground truth. On +nine real repositories the as-shipped traversal reached recall 0.022; the repaired version's +held-out point estimate is above grep's, but that it beats grep is not established +([refutation](research/empirical-refutation/)). The Node `forge impact` graph is a different, +regex-derived implementation and inherits none of these numbers. The production Forgekit equivalents are the Node implementations behind `forge preflight`, `forge route`, `forge atlas`, `forge impact`, and the MCP substrate tools. @@ -564,7 +600,12 @@ The production Forgekit equivalents are the Node implementations behind `forge p The [cognitive-substrate white paper](docs/cognitive-substrate/) explains the motivation, formal model, evidence map, adjacent work, and the relationship between the Python research prototypes and the current Node implementation. Treat the paper's measurements according to -their stated methodology; do not blend results from different codebases or evaluation sets. +their stated methodology; do not blend results from different codebases or evaluation sets. The +"faculty" and theorem language lives there and in [`research/`](research/), with dated +corrections: the papers now state what a frozen model does not guarantee (durable state across +calls, context beyond the window, weight updates from outcomes, reliable self-verification) rather +than claiming it cannot adapt at all, and they name the prior art (CoALA, Reflexion) the +architecture builds on. Their PDFs are historical, pre-correction editions; read the HTML. ## Public site @@ -588,6 +629,9 @@ not an AI application deployment. | --- | --- | | [`ONBOARDING.md`](ONBOARDING.md) | Five-minute setup and design principles | | [`docs/GUIDE.md`](docs/GUIDE.md) | Full command reference, worked examples, MCP schemas, and honest limits | +| [`docs/INTEGRATIONS.md`](docs/INTEGRATIONS.md) | Per-tool behaviour: config emission, MCP registration, automatic hooks, blocking — and how each is tested | +| [`docs/UNIVERSAL_ROUTING.md`](docs/UNIVERSAL_ROUTING.md) | The cross-provider router: model, shipped prior, evidence status and modeling limits | +| [`docs/status/`](docs/status/README.md) | Machine-readable status of every load-bearing claim (generated table) | | [`ARCHITECTURE.md`](ARCHITECTURE.md) | Runtime layers, emitters, state, and design decisions | | [`reports/benchmarks.md`](reports/benchmarks.md) | Reproducible benchmark snapshot, methodology, environment, and limitations | | [`research/python-prototypes/README.md`](research/python-prototypes/README.md) | Explicit maturity boundary for the Python research packages | diff --git a/ROADMAP.md b/ROADMAP.md index 25670832..14a10ff5 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -1,8 +1,9 @@ # Roadmap -forgekit is one brain for every AI coding agent — the cognitive substrate (memory, foresight, -guardrails) that a stateless model is missing, authored once and delivered as native config to -every tool. This is where that brain is headed. +forgekit is a beta toolkit for shared evidence-referenced memory, heuristic change-impact +analysis, and explicit verification around coding agents — the state and checks a model does not +carry between sessions on its own, authored once and delivered as native config to every tool. +This is where it is headed. Direction, not promises — shaped by the two field reports this project is grounded in (the SDLC pain-point map and the ecosystem landscape). Open a Discussion to weigh in. @@ -19,10 +20,13 @@ Gateway environments are supported end to end — `ANTHROPIC_AUTH_TOKEN` recogni `ANTHROPIC_BASE_URL`, and direct-HTTP LLM calls when the `claude` CLI is absent (`src/llm.js`). See [CHANGELOG.md](./CHANGELOG.md). -## Shipped — Substrate v2 (all phases P0–P8, v0.5.0) +## Shipped — Substrate v2 (P0–P3 and P5–P7, v0.5.0; P4 and P8 partial) The plan lives in [docs/plans/substrate-v2/](./docs/plans/substrate-v2/00-overview.md) -(phase dependency graph + acceptance gates, all marked done): every paper faculty and +(phase dependency graph + acceptance gates). Two phases do not meet their acceptance criteria and +are marked **partial**: P4 context assembly (budget overflow and pointer-only coverage were +reported as complete; the repair reports them) and P8 evaluation (no end-to-end cost has been +measured). Per-claim status is in [docs/status/](./docs/status/README.md). Every paper faculty and mechanism mapped to an algorithm, unified by the **Proof-Carrying Memory (PCM) protocol** (ADR-0006) — every stored thing is a claim that carries its evidence, earns confidence only from independent oracles, and merges across teammates conflict-free @@ -68,12 +72,17 @@ confidence only from independent oracles, and merges across teammates conflict-f ledger is the sole store. `FORGE_LEDGER_ONLY=0` is a one-release escape hatch that restores the file store. The only remaining step is deleting the now-dormant legacy write/read code once that escape hatch is removed in a later release. -- **OpenAI + Gemini provider detection** — extend `autoDetectProvider()` beyond - Anthropic/OpenRouter/LiteLLM (`OPENAI_API_KEY`, `GEMINI_API_KEY`) with the same - guided, low-configuration auto-detect contract. -- **Playwright loop** — still open: interaction checks and feeding verdicts back as - oracle evidence on design claims (fingerprinting itself shipped as - `forge uicheck visual`). +- **Evidence before stronger claims** — the 2026-09-26 review's evaluation plan: an impact study + on unseen repositories with the parser and relations frozen first, and co-edited-file prediction + and regression-test selection reported separately; the universal router on a new scaffold, time + split or language with real spend under an external cap, with a cascade cost model that + conditions on earlier failures (the in-repo replay under-predicts cascade cost by 5–22%); and + the P8 paired end-to-end cost harness. Until they run, the matching headlines stay `hypothesis` or `reported` in + [docs/status/](./docs/status/README.md). +- **UI checks stay advisory** until a labelled set (keyboard, focus order, zoom, overflow, + reduced motion, screen-reader names) measures their false-positive cost. OpenAI + Gemini + provider detection (0.17.0) and the Playwright interaction loop (`forge uicheck interact`, + 0.16.0) have shipped; this list used to show both as open. - **Advisory → gated promotions** — the measured-promotion gate has shipped (`src/promote.js`, generalizing the risk predictor's kill-criteria): a candidate only replaces a baseline when it beats it on held-out data, never by assertion. First diff --git a/bench/bench.mjs b/bench/bench.mjs index 1e8b6930..3ba4d12c 100644 --- a/bench/bench.mjs +++ b/bench/bench.mjs @@ -166,8 +166,32 @@ function fillLedger(dir, { count, offset = 0, evidenceEvery = 4 }) { } } -/** n in-memory artifact claims, each with one test.run confirm so it clears SERVE_FLOOR. */ -function makeArtifacts(n) { +/** + * A real git object to cite as evidence: a throwaway one-commit repo, and its full commit id, + * checked to resolve (`git cat-file -e`) — the same resolution appendEvidence performs. The + * old fixture cited `bench:artifact:`, an untyped ref that val() caps below SERVE_FLOOR, so + * every "exact"/"near" row actually measured a MISS (review F14). + * @returns {{dir: string, oid: string}} + */ +export function resolvedEvidenceRepo() { + const dir = mkdtempSync(join(tmpdir(), "forge-bench-git-")); + const g = (...args) => execFileSync("git", args, { cwd: dir, stdio: "ignore" }); + g("init"); + g("config", "user.email", "bench@example.invalid"); + g("config", "user.name", "bench"); + g("config", "commit.gpgsign", "false"); + writeFileSync(join(dir, "fixture.txt"), "reuse bench fixture\n"); + g("add", "-A"); + g("commit", "-m", "bench fixture"); + const oid = execFileSync("git", ["rev-parse", "HEAD"], { cwd: dir, encoding: "utf8" }).trim(); + g("cat-file", "-e", oid); // throws if it does not resolve + return { dir, oid }; +} + +/** n in-memory artifact claims, each with one RESOLVED test.run confirm (a `git:` object id) + * so it genuinely clears SERVE_FLOOR. `ref` overrides the evidence ref — a stale/unresolved + * fixture for the validation self-test. */ +export function makeArtifacts(n, { ref }) { const rand = mulberry32(42); const claims = []; const specs = []; @@ -179,23 +203,43 @@ function makeArtifacts(n) { 0, ); if (!minted.ok) throw new Error(minted.reason); - const o = outcomeRecord({ - oracle: "test.run", - result: "confirm", - ref: `bench:artifact:${i}`, - author: "bench", - t: 0, - }); - minted.claim.evidence = o.ok ? [o.outcome] : []; + const o = outcomeRecord({ oracle: "test.run", result: "confirm", ref, author: "bench", t: 0 }); + if (!o.ok) throw new Error(`bench evidence rejected: ${o.reason}`); + minted.claim.evidence = [o.outcome]; claims.push(minted.claim); } return { claims, specs }; } -const coldSketches = (claims) => { - for (const c of claims) delete c._sketch; +/** Strip EVERY memoized sketch the lookup path caches — ledger.js's claim-text `_sketch`/ + * `_terms` and reuse.js's `_specSketch`/`_keySketch` — so a "cold" row behaves like a fresh + * process. (It used to delete `_sketch` only, which reuse never reads.) */ +export const coldSketches = (claims) => { + for (const c of claims) { + delete c._sketch; + delete c._terms; + delete c._specSketch; + delete c._keySketch; + } }; +/** + * A row is only labeled with a tier it actually exercised: run the lookup once, untimed, and + * abort the benchmark unless the expected tier comes back (review F14). + * @param {any[]} claims + * @param {string} spec + * @param {"exact"|"near"|"adapt"|"miss"} tier + */ +export function expectTier(claims, spec, tier) { + coldSketches(claims); + const r = lookup(claims, spec); + if (r.tier !== tier) + throw new Error( + `bench fixture invalid: expected a ${tier} lookup, got ${r.tier}${r.reasons?.length ? ` (${r.reasons.slice(0, 2).join("; ")})` : ""}`, + ); + return r; +} + // --------------------------------------------------------------------------- // The run // --------------------------------------------------------------------------- @@ -204,6 +248,17 @@ function environment() { let commit = "unknown"; try { commit = execFileSync("git", ["rev-parse", "HEAD"], { cwd: REPO_ROOT }).toString().trim(); + // Numbers measured on an uncommitted tree are not the numbers OF that commit — say so. + const dirty = execFileSync("git", ["status", "--porcelain"], { cwd: REPO_ROOT }) + .toString() + .trim(); + if (dirty) commit += " + uncommitted changes"; + } catch {} + // The filesystem the tmpdir fixtures live on changes the disk-bound rows (A06). + let fsType = "unknown"; + try { + if (platform() === "linux") + fsType = execFileSync("stat", ["-f", "-c", "%T", tmpdir()], { encoding: "utf8" }).trim(); } catch {} return { node: process.version, @@ -212,6 +267,7 @@ function environment() { memGB: Math.round(totalmem() / 2 ** 30), platform: platform(), arch: arch(), + fsType, commit, date: new Date().toISOString(), }; @@ -240,7 +296,7 @@ function runBenchmarks() { "atlas", "full build (this repo)", tFull, - `${atlas.files} files, ${atlas.symbols.length} symbols, ${atlas.edges.length} edges`, + `${atlas.files} files, ${atlas.symbols.length} symbols, ${atlas.edges.length} edges${atlas.capped ? `, CAPPED at ${atlas.cap} files` : `, cap ${atlas.cap ?? "none"} not reached`}`, ); const tIncr = timeIt( () => { @@ -335,10 +391,17 @@ function runBenchmarks() { tFp, fmtRate(fpSpecs.length / (tFp.median / 1000)), ); + const evidence = resolvedEvidenceRepo(); + cleanup.push(evidence.dir); for (const n of [100, 1000]) { - const { claims, specs } = makeArtifacts(n); + const { claims, specs } = makeArtifacts(n, { ref: `git:${evidence.oid}` }); const exactQ = specs[n >> 1]; const nearQ = `${specs[n >> 1]} gently`; // superset tokens → Jaccard ≈ 0.97 → near tier + const missQ = "configure the blue ocean lighthouse keeper rotation schedule"; + // Validate BEFORE timing: each row must exercise the tier it is labeled with. + expectTier(claims, exactQ, "exact"); + expectTier(claims, nearQ, "near"); + expectTier(claims, missQ, "miss"); let hit = null; const tExact = timeIt( () => { @@ -347,7 +410,14 @@ function runBenchmarks() { }, { runs: 10, warmup: 2 }, ); - push("reuse", `lookup exact @ ${n} artifacts`, tExact, `tier=${hit.tier}`); + push("reuse", `lookup exact hit, cold @ ${n} artifacts`, tExact, `tier=${hit.tier}`); + const tExactWarm = timeIt( + () => { + hit = lookup(claims, exactQ); // memoized sketches kept: a long-lived process + }, + { runs: 10, warmup: 2 }, + ); + push("reuse", `lookup exact hit, warm @ ${n} artifacts`, tExactWarm, `tier=${hit.tier}`); const tNear = timeIt( () => { coldSketches(claims); @@ -357,10 +427,18 @@ function runBenchmarks() { ); push( "reuse", - `lookup near (LSH) @ ${n} artifacts`, + `lookup near hit (LSH), cold @ ${n} artifacts`, tNear, `tier=${hit.tier}, j=${hit.jaccard?.toFixed(2) ?? "-"}`, ); + const tMiss = timeIt( + () => { + coldSketches(claims); + hit = lookup(claims, missQ); + }, + { runs: 5, warmup: 1 }, + ); + push("reuse", `lookup miss, cold @ ${n} artifacts`, tMiss, `tier=${hit.tier}`); } // --- context: assemble() on this repo for a representative task ---------------- @@ -409,17 +487,27 @@ const RESULT_HEADERS = ["suite", "benchmark", "median", "p95", "runs", "notes"]; const QUALITY_HEADERS = ["case (target)", "precision", "recall", "F1", "predicted", "truth"]; const SERIES_HEADERS = ["series", "precision", "recall", "F1", "ground truth"]; -/** Paper prototype vs this repo — two different methodologies, side by side and - * labeled, NEVER averaged or blended. The paper row is a constant from the - * whitepaper (Figure 5 / deliverable-package.md), not something this harness ran. */ +/** Paper prototype vs this repo — different methodologies, side by side and labeled, + * NEVER averaged or blended. The two paper rows are constants, not something this harness + * ran: the self-built demo (whitepaper Figure 5 / deliverable-package.md), which the + * pre-registered field study REFUTED, and that field study's own result on real co-change + * data (research/empirical-refutation; recomputed by the 2026-09-26 external review). The + * regex atlas measured below is a different (Node) graph — not the evaluated Python oracle. */ export function seriesRows(quality) { return [ [ - "paper prototype (Python, mutation-derived)", + "paper prototype, self-built demo (REFUTED)", "0.63", "1.00", "0.75", - "mutation testing against a real suite", + "mutation testing on the authors' own fixture", + ], + [ + "paper prototype, field study (pooled, 9 repos)", + "0.40", + "0.02", + "0.04", + "759 files' mined co-change (research/empirical-refutation)", ], [ "this repo (regex atlas, hand-labeled)", @@ -475,7 +563,7 @@ function resultsMarkdown(env, rows, quality) { "", `Edited-file-only baseline recall over the same cases: **${quality.baseline.recall.toFixed(2)}**.`, "", - "Two methodologies, side by side — different codebases, different ground-truth", + "Different methodologies, side by side — different codebases, different ground-truth", "derivations, so the rows are comparable in spirit only and are never blended:", "", formatTable(SERIES_HEADERS, seriesRows(quality), { markdown: true }), diff --git a/bench/impact_cases.mjs b/bench/impact_cases.mjs index f0ae7195..c84ed07d 100644 --- a/bench/impact_cases.mjs +++ b/bench/impact_cases.mjs @@ -48,13 +48,15 @@ // - test/ledger.test.js imports { mergeStates } (:14) and calls it // (src/ledger_sync.js:3 also names it in the module header — same file, already labeled.) // -// claimText (src/ledger.js) — 10 files +// claimText (src/ledger.js) — 11 files // - src/ledger.js defines it (:610); sketchOf() (:636), termsOf() (:637) and :880 call it // - src/context.js imports { claimText } (:13) and calls it (:185) // - src/dash.js imports { claimText } (:16) and calls it (:58, :389, :400) // - src/deja.js imports { claimText } (:19) and calls it (:179) // - src/ledger_store.js imports { claimText } (:26) and calls it (:663) -// - src/cli.js dynamic-imports { claimText } (:874, :1644) and calls it +// - src/cli.js dynamic-imports { claimText } (:989) and calls it (:1012, :1023) +// - src/cli/memory.js dynamic-imports { claimText } (:251) and calls it (:257, :283, :313) +// — the recall/brain handlers moved here out of src/cli.js (review A03) // - src/cortex_mcp.js dynamic-imports { claimText } (:91) and calls it (:96, :106) // - test/ledger.test.js imports { claimText } (:8) and calls it // - src/learn_consolidate.js imports { claimText } (:32) and calls it (:110) @@ -123,6 +125,7 @@ export const IMPACT_CASES = [ "src/ledger_retention.js", "src/ledger_store.js", "src/cli.js", + "src/cli/memory.js", "src/cortex_mcp.js", "test/ledger.test.js", ], diff --git a/bench/universal-router/README.md b/bench/universal-router/README.md index 5883e2ae..ab40587d 100644 --- a/bench/universal-router/README.md +++ b/bench/universal-router/README.md @@ -4,13 +4,128 @@ outcomes (`forge route outcome`, then `forge route fit`). It was fitted from public per-task results: -- **Tasks:** the 500 issues of SWE-bench Verified (dataset revision `78f471b`); the issue text is the task. +- **Tasks:** the 500 issues of SWE-bench Verified (Hugging Face `SWE-bench/SWE-bench_Verified`, revision `78f471b`). The issue text (`problem_statement`) is the task. - **Runs:** eleven models from seven providers, each run once per issue with the same agent scaffold (mini-SWE-agent 2.0.0), taken from SWE-bench/experiments @ `40f164d` (runs dated 2026-02-17). Each run gives a verified resolved/unresolved outcome and the observed cost. -To regenerate it, build the input JSON, then run: +The fit chooses the latent dimension k and the prior scale by 3-fold cross-validation. That selection is recorded in `selection` inside the file. + +Raw per-task results are not redistributed here, only the fitted parameters and their provenance. +The scripts below download them from the pinned public sources into a work directory outside the +repository. Do not commit what they write. + +## Reproduction status (2026-09-26) + +| What | Status | Evidence | +|---|---|---| +| Source availability | Public and pinned | `sources.json` pins all twelve source files: URL, revision or commit, size and sha256, plus the git blob id of each GitHub file. Revision `78f471b` is in `SWE-bench/SWE-bench_Verified`; `princeton-nlp/SWE-bench_Verified` does not contain it. | +| Calculation: refit the shipped prior from public data | Reproduced exactly, from this repository alone | `reproduce.sh`, run from an empty work directory: all 176 values identical to `data/router_prior.json` (k=1, scale 4), with the router code of `d2abfa6` and again with the review changes later committed as `aedddf5` (runs below). Only `provenance.fittedAt` differs. | +| Pipeline: the run-4 held-out headline | Not reproducible from this repository | The 76.3% solved at $0.093 per task result comes from harness-bench. Its 150/350 split ids, pre-registration, baseline selection and metric aggregation are not in this repository. `holdout_eval.mjs` (below) is a different experiment and does not stand in for it. | +| Independent external replication | Repository-reported only | The 2026-09-22 replication at the end of this file is reported by this repository and has not been checked by a third party. The refits below ran in the project's own development environment: they show that the calculation reproduces, not that someone else replicated it. The external review of 2026-09-26 stopped its refit at a 180 s limit; the refit takes about 7 minutes on the machine below. | + +The refit runs of 2026-09-26, each from an empty work directory, on Node v22.22.2, Python 3.11.15, +pyarrow 25.0.1, Linux x64, Intel Xeon @ 2.80GHz (4 vCPU): + +| Router code the fit ran | Values of the shipped prior reproduced | Values only in the refit | Fit time | +|---|---|---|---| +| HEAD `d2abfa6`, no uncommitted change to `src/router`, `src/route.js` or `data/models.json` (`git status` before and after; router code unchanged since the fit) | 176 of 176, identical | 8: `provenance.build` | 413 s | +| HEAD `d2abfa6` plus uncommitted changes to `src/router/cost.js`, `index.js`, `policy.js` and `registry.js` (review F11/F12; 290 lines added, 41 removed), byte-identical to those files as later committed in `aedddf5` | 176 of 176, identical | 35: `provenance.build`, and the new cost diagnostics `method`, `s2Source`, `counts`, `alphaSE`, `excluded` | 424 s | + +## Reproduce the shipped prior + +```sh +sh bench/universal-router/reproduce.sh [WORK_DIR] +``` + +This is the one entry point. It needs network access to huggingface.co, raw.githubusercontent.com +and PyPI, `python3` 3.10 or newer with the `venv` module, and `node` 20 or newer. It writes only +under `WORK_DIR` (default `${TMPDIR:-/tmp}/forgekit-router-repro`): + +| Step | What runs | Writes | +|---|---|---| +| 1 | a virtual environment with the pinned pyarrow (`requirements.txt`) | `venv/` | +| 2 | `build_input.py`: downloads the files pinned in `sources.json`, checks each one's size and sha256 (and git blob id for the GitHub files), writes the fitting input | `cache/`, `universal_all.json` | +| 3 | `fit_prior.mjs`: refits the prior (single-threaded; minutes), timed | `router_prior.json`, `environment.txt` | +| 4 | `compare_priors.mjs`: checks that every value of `data/router_prior.json` is reproduced | `prior-comparison.json` | + +`environment.txt` records the Node, Python and pyarrow versions, the CPU, the fit time, and the +code the fit ran: the git HEAD and any uncommitted change to `src/router`, `src/route.js` or +`data/models.json`. + +It exits 0 when the refit reproduces every value of the shipped prior except +`provenance.fittedAt`, and 1 when any value differs or is missing. The comparison is exact: +numbers must be the same doubles. For each field (for example `mirt.L`) the report gives the +number of values that differ and the largest absolute difference. Values that only the refit has +are listed as added, not counted as a mismatch: `provenance.build` from the builder, and any field +newer code writes (such as the cost diagnostics in the status table). `compare_priors.mjs --strict` +also fails on additions. A rerun reads `cache/` instead of downloading, and +`build_input.py --offline --cache DIR` never downloads. + +What goes into the input, and why: + +- **Task text:** `problem_statement`, verbatim (252 statements keep their CRLF line endings). The builder reads only the `instance_id` and `problem_statement` columns. The parquet also holds the gold patch, the test patch and the evaluation fields, and none of them leave the builder. Keep that evaluator/agent boundary: `universal_all.json` carries outcome labels and costs, so it is training data for the fitter and must never be given to a coding agent as a prompt. +- **Order:** tasks by `instance_id` ascending (also the parquet's row order and the key order of every run file), models by registry id ascending. Bit-for-bit equality depends on both: feature means are summed in task order, and cross-validation folds are assigned by task index. +- **Models and runs:** `sources.json` maps each registry id to its run. The builder stops if `data/models.json` names a different `benchmark_run`. +- **Feature version:** the sha256 of `src/router/features.js` and of `src/route.js` (which holds the rubric and its exemplars), with CRLF read as LF. It is written to `source.build.featureVersion`, which a refit carries into `provenance.build`. +- **Zero-cost attempts:** Gemini 3 Flash's failed attempts on `django__django-15731` and `django__django-15814` are recorded at cost 0. They stay in the input as failures, which the ability model uses. The cost model uses only positive costs, so `cost.n` is 5,498 of 5,500. + +If a refit ever disagrees, look at `features` and `cost` first. Neither depends on the MIRT +optimiser, so a difference there points at the input: task text, order or registry prices. + +## Held-out replay in this repository (a new split, not run 4) + +`holdout_eval.mjs` is a new experiment on the same 500 tasks, with its own split. It does not +reproduce or test the run-4 headline, whose split and metric code are not here. + +```sh +node bench/universal-router/holdout_eval.mjs WORK_DIR/universal_all.json --out WORK_DIR/holdout.json +``` + +- **Split:** a Fisher-Yates shuffle of the ids in input order, driven by mulberry32 with seed 20260926 (`--seed`). The first 150 ids are dev and the other 350 are held out. Both id lists are in the output. +- **Fit:** `buildPrior(input, devIds)` on the 150 dev tasks only. The shipped prior, fitted on all 500 tasks, is never used. +- **Router:** `routeUniversal` with the dev fit, objective `match-best-single`, cascades of up to three models, over a registry restricted to the eleven fitted models. It sees only the task text. +- **Replay:** try the cascade's models in order until one's recorded attempt resolved the task. The cost is the sum of the recorded costs of the attempted models. +- **Baselines, chosen on dev only:** the model that solved the most dev tasks; the model with the lowest dev mean cost; and the cheapest fixed cascade of up to three models (1,111 considered) that solves at least as many dev tasks as that model, which is the router's own objective applied to dev outcomes. +- **Uncertainty:** paired bootstrap over held-out tasks, 10,000 seeded draws, 95% percentile intervals. +- **Record:** the output holds the seed, both id lists, the dev fit's selection table, every held-out decision, and the code the run loaded: the sha256 of each `src/router` file, `src/route.js` and `data/models.json`, the git HEAD and any uncommitted change. With a `target:` or `budget:` objective that no cascade can meet, the router returns its least-bad cascade as a fallback; the replay runs it and counts the task as infeasible (`match-best-single` is always feasible). + +Result for seed 20260926, run on 2026-09-26 (dev fit: k=1, scale 2, 58 s). The numbers are the +same with the router code of `d2abfa6` and with the review changes to `src/router` later +committed as `aedddf5` (see the status table). + +| Held-out, 350 tasks | Solved | Cost per task | Cost per solved task | +|---|---|---|---| +| Router | 80.0% | $0.124 | $0.155 | +| Best single model on dev: claude-opus-4.5 | 76.0% | $0.768 | $1.011 | +| Cheapest single model on dev: gpt-5-mini | 57.4% | $0.047 | $0.082 | +| Best fixed cascade on dev: minimax-m2.5 > gpt-5-mini > kimi-k2.5 | 80.6% | $0.136 | $0.169 | + +| Router minus | Solve rate, points [95% CI] | Cost per task [95% CI] | Tasks only one of the two solved (router, baseline) | +|---|---|---|---| +| Best single model | +4.0 [+0.6, +7.4] | -$0.644 [-$0.698, -$0.594] | 26, 12 | +| Cheapest single model | +22.6 [+18.3, +27.1] | +$0.077 [+$0.060, +$0.095] | 80, 1 | +| Best fixed cascade | -0.6 [-1.4, 0.0] | -$0.012 [-$0.024, -$0.003] | 0, 2 | + +What this shows: + +- Against the model that looked best on dev, the router solves more held-out tasks at about a sixth of the cost. Most of that comes from one cheap model: the router opened with minimax-m2.5 on 308 of the 350 tasks, and minimax-m2.5 alone solved 76.0% of the held-out tasks at $0.077 per task. +- Against a fixed cascade chosen on the same dev tasks, the router shows no advantage. It solved no task the cascade missed, the cascade solved two that it missed, and it cost $0.012 less per task. This agrees with `docs/UNIVERSAL_ROUTING.md`: with a 150-issue fit the router does not beat a fixed cascade chosen on the same dev data. +- Five more seeds (1 to 5), run after the result above as a sensitivity check, show the same picture. Against the dev-chosen fixed cascade, the router's difference ranged from -0.6 to +1.1 points and from -$0.012 to +$0.026 per task, and in no split was it both more accurate and cheaper. Against the best single model on dev, the result depends on which model won dev. When an Opus model won (3 of 6 splits), the router saved $0.46 to $0.64 per task and solved 2.9 to 4.0 points more, with an interval excluding zero only for seed 20260926. When minimax-m2.5 won (the other 3), the router solved 0.0 to 1.1 points more at $0.003 to $0.026 more per task. +- The chosen cascades' predicted success averaged 80.2% against 80.0% observed (across the six splits, predicted minus observed ranged from -2.8 to +9.1 points). Their expected cost was $0.102 per task against $0.124 replayed, and cost was under-predicted in all six splits, by 5% to 22% of the replayed cost. In these runs a failed attempt costs on average 1.2 to 2.0 times as much as a successful one, for every model, and a cascade reaches its second model only after a failure. The cost model does not condition on the outcome. + +Limits that apply to every number above: + +- One scaffold (mini-SWE-agent 2.0.0), one recorded attempt per model and task, twelve Python repositories (django alone is 231 of the 500 tasks), and the costs the runs recorded in February 2026, not today's prices. +- The replay assumes a perfect, free check between cascade attempts: it stops at the first attempt whose recorded SWE-bench label is resolved. A deployed cascade needs its own verifier, which can be wrong and costs money. +- The shipped prior is fitted on all 500 tasks, so it cannot be evaluated out of sample on them. `holdout_eval.mjs` refits on the dev tasks for that reason. +- One 150/350 split is small, and the six splits above differ by several points. +- SWE-bench Verified has known validity problems. OpenAI's analysis of 2026-02-23 found test flaws in 59.4% of an audited subset of 138 hard problems (not 59.4% of all 500 tasks) and reported evidence of contamination. Read these results as a historical comparison of routing policies on one benchmark, not as evidence of current coding ability or of present-day costs. + +## The original pipeline: harness-bench (run 4) + +The shipped prior and the run-4 held-out test were first built with harness-bench, which is not +part of this repository and is still the only way to rerun run 4: ```bash -# harness-bench writes the input from SWE-bench Verified + SWE-bench/experiments. # `build routing` makes the 150/350 split that `build universal` reads, so it runs first: # python3 -m hbench.cli build routing --swe-exp # python3 -m hbench.cli build universal --swe-exp --forgekit @@ -28,13 +143,12 @@ git -C swe-exp fetch --depth 1 --filter=blob:none origin 40f164d5b8f1d249bf95a6d git -C swe-exp checkout FETCH_HEAD ``` -The fit chooses the latent dimension k and the prior scale by 3-fold cross-validation. That selection is recorded in `selection` inside the file. - -Raw per-task results are not redistributed here, only the fitted parameters and their provenance. +`build_input.py` needs neither harness-bench nor a checkout, and the prior refitted from its input +matches the shipped one bit for bit. -## Independent replication (2026-09-22) +## Replication reported on 2026-09-22 -The run-4 benchmark and this prior were re-run from scratch on a second machine (Windows, Node 24, Python 3.12). The inputs were fetched as above, and the code was forgekit at `c5227db` (router code unchanged since the fit). +The run-4 benchmark and this prior were re-run from scratch on a second machine (Windows, Node 24, Python 3.12). The inputs were fetched as above, and the code was forgekit at `c5227db` (router code unchanged since the fit). This is the repository's own report; no third party has checked it. | What | Result | |---|---| @@ -44,6 +158,10 @@ The run-4 benchmark and this prior were re-run from scratch on a second machine | Headline | router 76.3% solved at $0.093 per task; best single model chosen on dev 75.1% at $0.364; pre-registered endpoint met | | Prior refit (`fit_prior.mjs`, all 500 issues) | all 176 fitted values identical to `data/router_prior.json` (k=1, scale 4); only `provenance.fittedAt` differs | +The `fitSeconds` values belong to the held-out test, which fits on the 150 dev issues. The refit +on all 500 issues takes several times longer: 413 s on the machine in the status table, where a +150-issue fit takes 58 s. + Two portability fixes were needed on Windows: - `fit_prior.mjs`'s default `--out` path (fixed here). - harness-bench's `adapters/forgekit_universal.mjs`, which passes plain paths to `import()` and so needs `pathToFileURL(...).href`. That file lives in harness-bench, not in this repo. diff --git a/bench/universal-router/build_input.py b/bench/universal-router/build_input.py new file mode 100644 index 00000000..4b3ec727 --- /dev/null +++ b/bench/universal-router/build_input.py @@ -0,0 +1,254 @@ +#!/usr/bin/env python3 +"""Build the universal router's fitting input from pinned public data. + +Reads the files pinned in bench/universal-router/sources.json (SWE-bench Verified as one parquet +file, and the eleven runs' per_instance_details.json from SWE-bench/experiments), checks every +file's size and sha256 (and, for GitHub files, the git blob id) before using it, and writes the +JSON that bench/universal-router/fit_prior.mjs reads: + + {source: {...}, tasks: [{id, text}], outcomes: {: {: {resolved, cost}}}} + +Evaluator/agent boundary: the parquet also holds the gold patch, the test patch and the +evaluation fields. Only two columns are read, instance_id and problem_statement, so nothing else +reaches the output. The output does carry outcome labels and costs: it is training data for the +router fit, never a prompt for a coding agent. + +Canonical orders. The fit is order-sensitive (feature means are summed in task order, and the +cross-validation folds are assigned by task index), so both orders are fixed here: + tasks instance_id ascending (code-point order) + models registry id ascending (code-point order) + +Attempts with a recorded cost of 0 (two failed Gemini runs at the pinned commit) are kept as +outcomes. fit_prior.mjs's cost model only uses positive costs, so it skips them. + +Usage: + python3 bench/universal-router/build_input.py --out universal_all.json [--cache DIR] [--offline] + +Needs Python 3.10+ and pyarrow (pinned in requirements.txt). Downloads honour HTTPS_PROXY. +""" + +import argparse +import hashlib +import io +import json +import math +import sys +import tempfile +import urllib.request +from pathlib import Path + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parent.parent +REL_HERE = HERE.relative_to(ROOT).as_posix() +# Files the text features depend on: the feature code, and the rubric (with its exemplars) that +# it calls. Hashed after normalising CRLF to LF, so a Windows checkout records the same value. +FEATURE_CODE = ["src/router/features.js", "src/route.js"] + + +def sha256_hex(data): + return hashlib.sha256(data).hexdigest() + + +def git_blob_id(data): + return hashlib.sha1(b"blob %d\0" % len(data) + data).hexdigest() + + +def problems_with(entry, data): + """Differences between a file's bytes and its pinned description (empty when it matches).""" + out = [] + if len(data) != entry["bytes"]: + out.append(f"size {len(data)} != pinned {entry['bytes']}") + got = sha256_hex(data) + if got != entry["sha256"]: + out.append(f"sha256 {got} != pinned {entry['sha256']}") + if "gitBlob" in entry and git_blob_id(data) != entry["gitBlob"]: + out.append(f"git blob {git_blob_id(data)} != pinned {entry['gitBlob']}") + return out + + +def fetch(entry, dest, offline): + """Return the verified bytes of one pinned file, reading `dest` or downloading into it.""" + if dest.exists(): + data = dest.read_bytes() + bad = problems_with(entry, data) + if not bad: + return data, "cached" + if offline: + sys.exit(f"build_input: {dest}: {'; '.join(bad)}") + print( + f" {dest.name}: cached copy does not match ({'; '.join(bad)}); downloading again" + ) + elif offline: + sys.exit(f"build_input: {dest} is missing and --offline was given") + request = urllib.request.Request( + entry["url"], headers={"User-Agent": "forgekit-build-input"} + ) + with urllib.request.urlopen(request, timeout=300) as response: + data = response.read() + bad = problems_with(entry, data) + if bad: + sys.exit(f"build_input: {entry['url']}: {'; '.join(bad)}") + dest.parent.mkdir(parents=True, exist_ok=True) + part = dest.with_name(dest.name + ".part") + part.write_bytes(data) + part.replace(dest) + return data, "downloaded" + + +def read_tasks(parquet_bytes): + """(instance_id, problem_statement) pairs. No other column is read.""" + import pyarrow.parquet as pq + + table = pq.read_table( + io.BytesIO(parquet_bytes), columns=["instance_id", "problem_statement"] + ) + ids = table.column("instance_id").to_pylist() + texts = table.column("problem_statement").to_pylist() + if len(set(ids)) != len(ids): + sys.exit("build_input: duplicate instance_id in the dataset") + for tid, text in zip(ids, texts): + if not isinstance(tid, str) or not isinstance(text, str) or not text: + sys.exit(f"build_input: {tid!r}: missing instance_id or problem_statement") + return ids, dict(zip(ids, texts)) + + +def read_outcomes(model_id, raw, task_ids): + """{instance_id: {resolved, cost}} in canonical task order, after strict validation.""" + per = json.loads(raw) + if not isinstance(per, dict): + sys.exit(f"build_input: {model_id}: per_instance_details.json is not an object") + missing = sorted(set(task_ids) - set(per)) + extra = sorted(set(per) - set(task_ids)) + if missing or extra: + sys.exit( + f"build_input: {model_id}: {len(missing)} dataset tasks missing, " + f"{len(extra)} unknown ids (e.g. {(missing + extra)[:3]})" + ) + rows = {} + for tid in task_ids: + rec = per[tid] + resolved, cost = rec.get("resolved"), rec.get("cost") + if not isinstance(resolved, bool): + sys.exit( + f"build_input: {model_id} {tid}: resolved is {resolved!r}, not a boolean" + ) + numeric = isinstance(cost, (int, float)) and not isinstance(cost, bool) + if not numeric or not math.isfinite(cost) or cost < 0: + sys.exit( + f"build_input: {model_id} {tid}: cost is {cost!r}, not a finite number >= 0" + ) + rows[tid] = {"resolved": resolved, "cost": cost} + return rows + + +def feature_version(): + out = {} + for rel in FEATURE_CODE: + data = (ROOT / rel).read_bytes().replace(b"\r\n", b"\n") + out[rel] = f"sha256:{sha256_hex(data)}" + return out + + +def main(): + ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + ap.add_argument( + "--out", required=True, help="where to write the fitting input JSON" + ) + ap.add_argument( + "--cache", + help="directory that holds (or receives) the pinned source files; " + "without it they are downloaded into a temporary directory", + ) + ap.add_argument( + "--offline", action="store_true", help="never download; use --cache only" + ) + args = ap.parse_args() + if args.offline and not args.cache: + sys.exit("build_input: --offline needs --cache") + + sources_path = HERE / "sources.json" + sources = json.loads(sources_path.read_text(encoding="utf-8")) + ds, ex = sources["dataset"], sources["experiments"] + runs = ex["runs"] + model_ids = sorted(runs) + + # The registry is what fit_prior.mjs keeps models by: a pinned run whose id is not in it + # would be dropped silently, and a registry that names another run would be misleading. + registry = json.loads((ROOT / "data" / "models.json").read_text(encoding="utf-8")) + by_id = {m["id"]: m for m in registry["models"]} + for mid in model_ids: + if mid not in by_id: + sys.exit( + f"build_input: {mid} is pinned in sources.json but not in data/models.json" + ) + if by_id[mid].get("benchmark_run") != runs[mid]["run"]: + sys.exit( + f"build_input: {mid}: data/models.json says benchmark_run " + f"{by_id[mid].get('benchmark_run')!r}, sources.json pins {runs[mid]['run']!r}" + ) + + with tempfile.TemporaryDirectory(prefix="router-input-") as tmp: + cache = Path(args.cache) if args.cache else Path(tmp) + print(f"sources: {sources_path.relative_to(ROOT).as_posix()} (cache: {cache})") + dest = cache / ds["repo"].replace("/", "__") / ds["revision"] / ds["path"] + parquet, how = fetch(ds, dest, args.offline) + print(f" {ds['repo']}@{ds['revision'][:7]} {ds['path']}: {how}, sha256 ok") + task_ids, texts = read_tasks(parquet) + task_ids = sorted(task_ids) + exp_repo = ex["repo"].removeprefix("https://github.com/") + outcomes = {} + for mid in model_ids: + run = runs[mid] + dest = cache / exp_repo.replace("/", "__") / ex["commit"] / run["path"] + raw, how = fetch(run, dest, args.offline) + print(f" {mid:<18} {run['run']}: {how}, sha256 + git blob ok") + outcomes[mid] = read_outcomes(mid, raw, task_ids) + + source = { + "benchmark": f"{ds['name']} (rev {ds['revision'][:7]})", + "runs": f"{exp_repo}@{ex['commit'][:7]}, {ex['scaffold']}, dated {ex['runDate']}", + "split": "all", + "build": { + "builder": f"{REL_HERE}/build_input.py", + "dataset": f"{ds['repo']}@{ds['revision']}", + "experiments": f"{exp_repo}@{ex['commit']}", + "taskText": "problem_statement, verbatim", + "taskOrder": "instance_id ascending", + "modelOrder": "registry id ascending", + "featureVersion": feature_version(), + }, + } + doc = { + "source": source, + "tasks": [{"id": tid, "text": texts[tid]} for tid in task_ids], + "outcomes": outcomes, + } + # Bytes, not text mode: the same file (and sha256) on every platform. + text = json.dumps(doc, ensure_ascii=False, separators=(",", ":")) + "\n" + out = Path(args.out) + out.parent.mkdir(parents=True, exist_ok=True) + part = out.with_name(out.name + ".part") + part.write_bytes(text.encode("utf-8")) + part.replace(out) + + cells = [(mid, tid, o) for mid in model_ids for tid, o in outcomes[mid].items()] + zero = [(mid, tid, o["resolved"]) for mid, tid, o in cells if o["cost"] == 0] + print(f"tasks: {len(task_ids)} models: {len(model_ids)} outcomes: {len(cells)}") + for mid in model_ids: + solved = sum(o["resolved"] for o in outcomes[mid].values()) + print(f" {mid:<18} resolved {solved}/{len(task_ids)}") + print( + f"zero-cost attempts kept as outcomes: {len(zero)} " + f"({sum(1 for z in zero if not z[2])} unresolved); the cost fit skips them" + ) + for mid, tid, resolved in zero: + print(f" {mid} {tid} resolved={str(resolved).lower()} cost=0") + for rel, digest in source["build"]["featureVersion"].items(): + print(f"feature code {rel}: {digest}") + print( + f"wrote {out} ({len(text.encode('utf-8'))} bytes, sha256 {sha256_hex(text.encode('utf-8'))})" + ) + + +if __name__ == "__main__": + main() diff --git a/bench/universal-router/compare_priors.mjs b/bench/universal-router/compare_priors.mjs new file mode 100644 index 00000000..df9f8374 --- /dev/null +++ b/bench/universal-router/compare_priors.mjs @@ -0,0 +1,169 @@ +#!/usr/bin/env node +// Compare two router prior files (the data/router_prior.json format) value by value. +// +// node bench/universal-router/compare_priors.mjs +// [--ignore ]... [--strict] +// +// The question it answers: does reproduce every value of ? Each leaf of +// must exist in with an identical value: numbers with ===, which after JSON +// parsing is bit-for-bit equality of the doubles, and strings, booleans and nulls likewise. +// provenance.fittedAt is always skipped; each --ignore adds a dotted path (a prefix skips its +// whole subtree), and the report lists every path it skipped. +// +// Leaves that exist only in are reported as `added`, grouped by field. Newer code can +// write fields the expected file predates (build metadata in provenance.build, diagnostics in +// cost such as cost.alphaSE), and an addition does not change any expected value, so additions +// are not a mismatch unless --strict is given. +// +// Output: a JSON report on stdout (leaf counts; per field, meaning the path without array +// indices such as mirt.L, the number of differing values and the maximum absolute difference; +// examples; missing and added paths) and a one-line summary on stderr. Exit code 0 when every +// compared value is identical (and, with --strict, nothing was added), 1 otherwise, 2 on a +// usage error. +import { readFileSync } from "node:fs"; + +const args = process.argv.slice(2); +const files = []; +const ignore = ["provenance.fittedAt"]; +let strict = false; +let badFlag = false; +for (let i = 0; i < args.length; i++) { + if (args[i] === "--ignore" && args[i + 1]) ignore.push(args[++i]); + else if (args[i] === "--strict") strict = true; + else if (args[i].startsWith("--")) badFlag = true; + else files.push(args[i]); +} +if (badFlag || files.length !== 2) { + console.error( + "usage: compare_priors.mjs [--ignore ]... [--strict]", + ); + process.exit(2); +} +// An unreadable file is a usage error (2), never a mismatch (1). +const load = (f) => { + try { + return JSON.parse(readFileSync(f, "utf8")); + } catch (e) { + console.error(`compare_priors: cannot read ${f}: ${e.message}`); + process.exit(2); + } +}; +const [expected, actual] = files.map(load); + +/** + * Leaves of a JSON value: key = JSON of the full path, with its display path and its field + * (the path without array indices). An empty object or array is a leaf of its own. + * @returns {Map} + */ +function leaves(value, path = [], field = [], acc = new Map()) { + if (value !== null && typeof value === "object") { + const isArray = Array.isArray(value); + const keys = isArray ? value.map((_, i) => i) : Object.keys(value); + if (!keys.length) + acc.set(JSON.stringify(path), { path: path.join("."), field: field.join("."), value }); + for (const k of keys) leaves(value[k], [...path, k], isArray ? field : [...field, k], acc); + } else acc.set(JSON.stringify(path), { path: path.join("."), field: field.join("."), value }); + return acc; +} + +const skipped = (p) => ignore.some((g) => p === g || p.startsWith(`${g}.`)); +const same = (x, y) => + x === y || + (typeof x === "object" && typeof y === "object" && JSON.stringify(x) === JSON.stringify(y)); + +const A = leaves(expected); +const B = leaves(actual); +const fields = new Map(); +const missing = []; +const added = []; +const ignoredPaths = []; +let compared = 0; +let identical = 0; +for (const [key, a] of A) { + if (skipped(a.path)) { + ignoredPaths.push(a.path); + continue; + } + const b = B.get(key); + if (!b) { + missing.push(a.path); + continue; + } + compared++; + if (!fields.has(a.field)) + fields.set(a.field, { values: 0, differing: 0, maxAbsDiff: 0, nonNumeric: 0, examples: [] }); + const f = fields.get(a.field); + f.values++; + if (same(a.value, b.value)) { + identical++; + continue; + } + f.differing++; + if (typeof a.value === "number" && typeof b.value === "number") + f.maxAbsDiff = Math.max(f.maxAbsDiff, Math.abs(a.value - b.value)); + else f.nonNumeric++; + if (f.examples.length < 5) f.examples.push({ path: a.path, expected: a.value, actual: b.value }); +} +const addedByField = new Map(); +for (const [key, b] of B) { + if (A.has(key)) continue; + if (skipped(b.path)) ignoredPaths.push(b.path); + else { + added.push(b.path); + addedByField.set(b.field, (addedByField.get(b.field) ?? 0) + 1); + } +} + +const differing = compared - identical; +const exactMatch = differing === 0 && !missing.length; +const pass = exactMatch && !(strict && added.length); +const report = { + expected: files[0], + actual: files[1], + exactMatch, + strict, + pass, + ignored: { patterns: ignore, paths: [...new Set(ignoredPaths)] }, + leaves: { compared, identical, differing, missing: missing.length, added: added.length }, + fields: Object.fromEntries( + [...fields].map(([name, f]) => [ + name, + { + values: f.values, + differing: f.differing, + maxAbsDiff: f.differing > f.nonNumeric ? f.maxAbsDiff : f.differing ? null : 0, + ...(f.nonNumeric ? { nonNumericDiffering: f.nonNumeric } : {}), + ...(f.examples.length ? { examples: f.examples } : {}), + }, + ]), + ), + missing, + added: { byField: Object.fromEntries(addedByField), paths: added }, +}; +console.log(JSON.stringify(report, null, 2)); + +const parts = exactMatch + ? [`exact match: ${identical}/${compared} values of ${files[0]} identical`] + : [ + `MISMATCH: ${differing}/${compared} values differ`, + ...(missing.length ? [`${missing.length} paths missing from ${files[1]}`] : []), + ...[...fields] + .filter(([, f]) => f.differing) + .map(([name, f]) => + f.differing > f.nonNumeric + ? `${name} ${f.differing}/${f.values}, max |diff| ${f.maxAbsDiff.toPrecision(3)}` + : `${name} ${f.differing}/${f.values} (text)`, + ), + ]; +if (added.length) { + // Group additions by their first two path segments: "cost.alphaSE 11, provenance.build 10". + const groups = new Map(); + for (const [field, count] of addedByField) { + const g = field.split(".").slice(0, 2).join("."); + groups.set(g, (groups.get(g) ?? 0) + count); + } + const list = [...groups].map(([g, c]) => `${g} ${c}`).join(", "); + parts.push(`${added.length} values only in ${files[1]}${strict ? " (--strict)" : ""}: ${list}`); +} +console.error(`${parts.join("; ")} (skipped: ${ignore.join(", ")})`); +process.exit(pass ? 0 : 1); diff --git a/bench/universal-router/holdout_eval.mjs b/bench/universal-router/holdout_eval.mjs new file mode 100644 index 00000000..a21d5c38 --- /dev/null +++ b/bench/universal-router/holdout_eval.mjs @@ -0,0 +1,455 @@ +#!/usr/bin/env node +// Held-out replay of the universal router on public SWE-bench Verified outcomes (review E3, +// stage 1), run entirely from this repository. +// +// This is NOT harness-bench run 4. That experiment's 150/350 split ids, pre-registration and +// metric code live in harness-bench, not here, so its headline (76.3% solved at $0.093 per task) +// is neither reproduced nor tested by this script. This is a separate experiment with its own +// seeded split of the same 500 tasks. +// +// node bench/universal-router/holdout_eval.mjs --out +// [--seed 20260926] [--dev 150] [--draws 10000] [--objective match-best-single] [--max-depth 3] +// +// is what build_input.py writes. The steps: +// 1. Split. Shuffle the task ids (input order: instance_id ascending) with a Fisher-Yates +// shuffle driven by mulberry32(seed). The first --dev ids are dev, the rest are held out. +// Both id lists go into the output. +// 2. Fit on dev only: buildPrior(input, devIds). The shipped all-500 prior is never used. +// 3. Baselines, each chosen on dev outcomes only: the single model that solved the most dev +// tasks; the single model with the lowest dev mean cost; and the fixed cascade (up to +// --max-depth models) with the lowest dev mean cost among those that solve at least as many +// dev tasks as that best single model (the router's match-best-single objective applied to +// dev outcomes). +// 4. Replay each held-out task. routeUniversal() picks a cascade with the dev fit, over a +// registry restricted to the fitted models. The cascade, and each baseline, is then played +// on the recorded outcomes: try the models in order until one resolved the task. Its cost +// is the sum of the recorded costs of the attempted models. A failed attempt recorded at +// cost 0 costs 0 and is counted. +// 5. Paired bootstrap over held-out tasks (seeded): 95% percentile intervals for the router's +// solve-rate and mean-cost differences to each baseline. +// +// The replay assumes a perfect check between cascade attempts (the recorded SWE-bench label) +// that costs nothing. A real cascade needs its own verifier, which can be wrong and costs money. +import { execFileSync } from "node:child_process"; +import { createHash } from "node:crypto"; +import { readdirSync, readFileSync, writeFileSync } from "node:fs"; +import { arch, cpus, platform } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { routeUniversal } from "../../src/router/index.js"; +import { cascades } from "../../src/router/policy.js"; +import { buildPrior } from "../../src/router/prior.js"; +import { loadRegistry } from "../../src/router/registry.js"; + +const args = process.argv.slice(2); +const opt = (name, fallback) => (args.includes(name) ? args[args.indexOf(name) + 1] : fallback); +const inputPath = args[0]; +const outPath = opt("--out"); +if (!inputPath || inputPath.startsWith("--") || !outPath) { + console.error("usage: holdout_eval.mjs --out [--seed N] [--dev N]"); + process.exit(2); +} +const seed = Number(opt("--seed", 20260926)) >>> 0; +const nDev = Number(opt("--dev", 150)); +const draws = Number(opt("--draws", 10000)); +const objective = opt("--objective", "match-best-single"); +const maxDepth = Number(opt("--max-depth", 3)); + +const t0 = Date.now(); + +// The code this run loaded, recorded before anything else happens: the router's source files +// and the rubric (sha256, CRLF read as LF), the git HEAD, and any uncommitted change to them. +const ROOT = fileURLToPath(new URL("../../", import.meta.url)); +function codeState() { + const files = [ + ...readdirSync(join(ROOT, "src/router")) + .filter((f) => f.endsWith(".js")) + .sort() + .map((f) => `src/router/${f}`), + "src/route.js", + "data/models.json", + ]; + const sha256 = Object.fromEntries( + files.map((f) => { + const text = readFileSync(join(ROOT, f), "utf8").replace(/\r\n/g, "\n"); + return [f, `sha256:${createHash("sha256").update(text).digest("hex")}`]; + }), + ); + const git = (...a) => + execFileSync("git", ["-C", ROOT, ...a], { + encoding: "utf8", + stdio: ["ignore", "pipe", "ignore"], + }).trim(); + let gitHead = null; + let uncommitted = null; + try { + gitHead = git("rev-parse", "HEAD"); + uncommitted = + git("diff", "--stat", "HEAD", "--", "src/router", "src/route.js", "data/models.json") || + "none"; + } catch {} + return { gitHead, uncommitted, sha256 }; +} +const code = codeState(); +const rawInput = readFileSync(inputPath); +const input = JSON.parse(rawInput.toString("utf8")); +const taskIds = input.tasks.map((t) => t.id); +const textOf = new Map(input.tasks.map((t) => [t.id, t.text])); +if (!(Number.isInteger(nDev) && nDev > 0 && nDev < taskIds.length)) + throw new Error(`--dev must be an integer in 1..${taskIds.length - 1}`); +if (!(Number.isInteger(draws) && draws > 0)) throw new Error("--draws must be a positive integer"); +if (!(Number.isInteger(maxDepth) && maxDepth > 0)) + throw new Error("--max-depth must be a positive integer"); + +/** mulberry32: a small, well-known 32-bit PRNG; one seed gives one stream on every platform. */ +function mulberry32(a) { + let s = a >>> 0; + return () => { + s = (s + 0x6d2b79f5) | 0; + let t = Math.imul(s ^ (s >>> 15), 1 | s); + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; +} + +// 1. Split. +const shuffled = taskIds.slice(); +const splitRand = mulberry32(seed); +for (let i = shuffled.length - 1; i > 0; i--) { + const j = Math.floor(splitRand() * (i + 1)); + [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]]; +} +const devSet = new Set(shuffled.slice(0, nDev)); +const dev = taskIds.filter((id) => devSet.has(id)); +const heldOut = taskIds.filter((id) => !devSet.has(id)); + +// 2. Fit on dev only. +const registry = loadRegistry(null); +const models = Object.keys(input.outcomes).filter((id) => registry.models.some((m) => m.id === id)); +const restricted = { + models: registry.models.filter((m) => models.includes(m.id)), + sources: registry.sources, +}; +const tFit = Date.now(); +const fit = buildPrior(input, devSet); +const fitSeconds = (Date.now() - tFit) / 1000; +if (fit.provenance.tasks !== dev.length) throw new Error("dev fit saw tasks outside the dev set"); + +// Play a model sequence on one task's recorded outcomes. +function play(seq, id) { + let cost = 0; + let zeroCostFailed = 0; + for (let i = 0; i < seq.length; i++) { + const o = input.outcomes[seq[i]]?.[id]; + if (!o || !Number.isFinite(o.cost)) + throw new Error(`no recorded outcome for ${seq[i]} on ${id}`); + cost += o.cost; + if (o.resolved) return { solved: 1, cost, attempts: i + 1, zeroCostFailed }; + if (o.cost === 0) zeroCostFailed++; + } + return { solved: 0, cost, attempts: seq.length, zeroCostFailed }; +} + +function summarize(rows) { + const n = rows.length; + const sum = (k) => rows.reduce((s, r) => s + r[k], 0); + const solved = sum("solved"); + const totalCost = sum("cost"); + return { + tasks: n, + solved, + solveRate: solved / n, + meanCost: totalCost / n, + costPerSolved: solved ? totalCost / solved : null, + totalCost, + meanAttempts: sum("attempts") / n, + zeroCostFailedAttempts: sum("zeroCostFailed"), + }; +} +const playAll = (seq, ids) => ids.map((id) => play(seq, id)); +const bySeq = (a, b) => (a.seq.join(" ") < b.seq.join(" ") ? -1 : 1); + +// 3. Baselines, chosen on dev outcomes only. +const devSingles = models.map((m) => ({ seq: [m], ...summarize(playAll([m], dev)) })); +const bestSingle = devSingles + .slice() + .sort((a, b) => b.solved - a.solved || a.totalCost - b.totalCost || bySeq(a, b))[0]; +const cheapestSingle = devSingles + .slice() + .sort((a, b) => a.totalCost - b.totalCost || b.solved - a.solved || bySeq(a, b))[0]; +let bestFixedCascade = null; +let cascadesConsidered = 0; +for (const seq of cascades(models, maxDepth)) { + cascadesConsidered++; + const s = { seq, ...summarize(playAll(seq, dev)) }; + if (s.solved < bestSingle.solved) continue; + const c = bestFixedCascade; + if ( + !c || + s.totalCost < c.totalCost || + (s.totalCost === c.totalCost && + (s.solved > c.solved || (s.solved === c.solved && seq.length < c.seq.length))) + ) + bestFixedCascade = s; +} + +// 4. Router replay on held-out tasks. +const decisions = heldOut.map((id) => { + const r = routeUniversal(null, textOf.get(id), { + model: fit, + registry: restricted, + objective, + maxDepth, + }); + // A target or budget no cascade can meet comes back as ok:false with the least-bad cascade + // as `fallback` (match-best-single is always feasible). The replay runs that fallback and + // counts the task as infeasible. + const pick = r.ok ? r : r.fallback; + if (!pick?.cascade) throw new Error(`router gave no cascade for ${id}: ${r.reason}`); + const seq = pick.cascade.map((c) => c.model); + return { + id, + cascade: seq, + feasible: Boolean(r.ok), + pSuccess: pick.pSuccess, + expectedCost: pick.expectedCost, + ...play(seq, id), + }; +}); +const infeasible = decisions.filter((d) => !d.feasible).length; + +const policies = { + router: decisions, + bestSingle: playAll(bestSingle.seq, heldOut), + cheapestSingle: playAll(cheapestSingle.seq, heldOut), + bestFixedCascade: playAll(bestFixedCascade.seq, heldOut), +}; +const baselines = ["bestSingle", "cheapestSingle", "bestFixedCascade"]; + +// 5. Paired bootstrap over held-out tasks. +const bootSeed = (seed ^ 0x9e3779b9) >>> 0; +const bootRand = mulberry32(bootSeed); +const n = heldOut.length; +const cols = Object.fromEntries( + Object.entries(policies).map(([k, rows]) => [ + k, + { + solved: Float64Array.from(rows, (r) => r.solved), + cost: Float64Array.from(rows, (r) => r.cost), + }, + ]), +); +const samples = Object.fromEntries( + baselines.map((b) => [ + b, + { + dSolve: new Float64Array(draws), + dCost: new Float64Array(draws), + ratio: new Float64Array(draws), + }, + ]), +); +const names = Object.keys(policies); +const sums = Object.fromEntries(names.map((k) => [k, { solved: 0, cost: 0 }])); +for (let d = 0; d < draws; d++) { + for (const k of names) { + sums[k].solved = 0; + sums[k].cost = 0; + } + for (let i = 0; i < n; i++) { + const j = Math.floor(bootRand() * n); + for (const k of names) { + sums[k].solved += cols[k].solved[j]; + sums[k].cost += cols[k].cost[j]; + } + } + for (const b of baselines) { + samples[b].dSolve[d] = (sums.router.solved - sums[b].solved) / n; + samples[b].dCost[d] = (sums.router.cost - sums[b].cost) / n; + samples[b].ratio[d] = sums.router.cost / sums[b].cost; + } +} +/** Percentile with linear interpolation between order statistics (numpy's default). */ +function quantile(sorted, q) { + const h = (sorted.length - 1) * q; + const lo = Math.floor(h); + return sorted[lo] + (h - lo) * (sorted[Math.min(lo + 1, sorted.length - 1)] - sorted[lo]); +} +const interval = (estimate, arr) => { + const s = Float64Array.from(arr).sort(); + return { estimate, ci95: [quantile(s, 0.025), quantile(s, 0.975)] }; +}; +const heldOutSummary = Object.fromEntries(names.map((k) => [k, summarize(policies[k])])); +const comparisons = Object.fromEntries( + baselines.map((b) => { + const r = heldOutSummary.router; + const x = heldOutSummary[b]; + // Paired discordant tasks: with only a handful, the bootstrap interval says little. + let routerOnly = 0; + let baselineOnly = 0; + for (let i = 0; i < n; i++) { + if (cols.router.solved[i] > cols[b].solved[i]) routerOnly++; + else if (cols.router.solved[i] < cols[b].solved[i]) baselineOnly++; + } + return [ + b, + { + solveRateDelta: interval(r.solveRate - x.solveRate, samples[b].dSolve), + meanCostDelta: interval(r.meanCost - x.meanCost, samples[b].dCost), + meanCostRatio: interval(r.meanCost / x.meanCost, samples[b].ratio), + solvedByOnlyOne: { router: routerOnly, baseline: baselineOnly }, + }, + ]; + }), +); + +// Calibration of the router's own predictions for the cascades it chose (quintiles of pSuccess). +const byP = decisions.slice().sort((a, b) => a.pSuccess - b.pSuccess); +const mean = (rows, f) => rows.reduce((s, r) => s + f(r), 0) / rows.length; +const calibration = { + meanPredictedSuccess: mean(decisions, (r) => r.pSuccess), + observedSolveRate: heldOutSummary.router.solveRate, + brier: mean(decisions, (r) => (r.pSuccess - r.solved) ** 2), + meanExpectedCost: mean(decisions, (r) => r.expectedCost), + meanObservedCost: heldOutSummary.router.meanCost, + quintiles: [0, 1, 2, 3, 4].map((q) => { + const rows = byP.slice(Math.round((q * n) / 5), Math.round(((q + 1) * n) / 5)); + return { + tasks: rows.length, + predicted: mean(rows, (r) => r.pSuccess), + observed: mean(rows, (r) => r.solved), + expectedCost: mean(rows, (r) => r.expectedCost), + observedCost: mean(rows, (r) => r.cost), + }; + }), +}; + +// What the router chose. +const tally = (f) => { + const m = new Map(); + for (const r of decisions) m.set(f(r), (m.get(f(r)) ?? 0) + 1); + return Object.fromEntries([...m].sort((a, b) => b[1] - a[1])); +}; +const composition = { + cascadeLength: tally((r) => r.cascade.length), + firstModel: tally((r) => r.cascade[0]), + cascades: tally((r) => r.cascade.join(" > ")), +}; + +// Per-repository slices (held-out only; small slices are noisy). +const repoOf = (id) => id.split("__")[0]; +const repos = [...new Set(heldOut.map(repoOf))].sort(); +const perRepository = repos.map((repo) => { + const idx = heldOut.map((id, i) => (repoOf(id) === repo ? i : -1)).filter((i) => i >= 0); + const slice = (k) => { + const s = summarize(idx.map((i) => policies[k][i])); + return { solveRate: s.solveRate, meanCost: s.meanCost }; + }; + return { repo, tasks: idx.length, ...Object.fromEntries(names.map((k) => [k, slice(k)])) }; +}); + +const cpu = cpus(); +const result = { + experiment: + "holdout_eval.mjs: in-repo held-out replay of the universal router on SWE-bench Verified " + + "outcomes (review E3, stage 1)", + notTheHeadline: + "A new seeded split made by this script. It is not harness-bench run 4, whose split ids, " + + "pre-registration and metric code are not in this repository, so it neither reproduces " + + "nor tests the 76.3% / $0.093 headline.", + assumptions: [ + "Each model's single recorded attempt per task (mini-SWE-agent 2.0.0, February 2026 costs) " + + "stands for any attempt of that model on that task.", + "A cascade stops at the first attempt whose recorded SWE-bench label is resolved: a perfect " + + "check between attempts, with no verification cost.", + "The router sees only the task text (problem_statement); labels and costs are used only " + + "to fit on dev and to score held-out decisions.", + ], + generatedAt: new Date().toISOString(), + environment: { + node: process.version, + platform: `${platform()}-${arch()}`, + cpu: cpu[0]?.model ?? null, + cpus: cpu.length, + }, + code, + input: { + path: inputPath, + sha256: createHash("sha256").update(rawInput).digest("hex"), + source: input.source ?? null, + tasks: taskIds.length, + models, + }, + split: { + seed, + prng: "mulberry32", + method: + "Fisher-Yates shuffle of the task ids in input order (instance_id ascending); " + + "the first `dev` ids are dev, the rest held out. Id lists are in input order.", + dev: { tasks: dev.length, ids: dev }, + heldOut: { tasks: heldOut.length, ids: heldOut }, + }, + fit: { + seconds: fitSeconds, + tasks: fit.provenance.tasks, + outcomes: fit.provenance.outcomes, + k: fit.mirt.k, + scale: fit.selection.chosen.scale, + selection: fit.selection, + }, + policy: { objective, maxDepth, candidates: models.length, infeasibleTasks: infeasible }, + devSelection: { + singles: devSingles, + bestSingle: bestSingle.seq, + cheapestSingle: cheapestSingle.seq, + bestFixedCascade: bestFixedCascade.seq, + bestFixedCascadeDev: bestFixedCascade, + cascadesConsidered, + }, + heldOut: { + ...heldOutSummary, + singles: Object.fromEntries(models.map((m) => [m, summarize(playAll([m], heldOut))])), + }, + bootstrap: { + draws, + seed: bootSeed, + method: "paired: resample held-out tasks with replacement; percentile 95% intervals", + router_minus: comparisons, + }, + calibration, + composition, + perRepository, + perTask: decisions, + elapsedSeconds: (Date.now() - t0) / 1000, +}; +writeFileSync(outPath, `${JSON.stringify(result, null, 2)}\n`); + +const pct = (v) => `${(100 * v).toFixed(1)}%`; +const usd = (v) => `$${v.toFixed(3)}`; +const line = (name, s, seq) => + ` ${name.padEnd(20)} ${pct(s.solveRate).padStart(6)} solved ${usd(s.meanCost)}/task ` + + `${s.costPerSolved === null ? "-" : usd(s.costPerSolved)}/solved${seq ? ` [${seq.join(" > ")}]` : ""}`; +console.log( + `holdout_eval (NOT the harness-bench split): seed ${seed}, dev ${dev.length} / held-out ${n}, ` + + `fit k=${fit.mirt.k} scale=${fit.selection.chosen.scale} in ${fitSeconds.toFixed(1)}s`, +); +if (infeasible) + console.log( + ` ${objective}: infeasible on ${infeasible} tasks (their fallback cascade was replayed)`, + ); +console.log(line("router", heldOutSummary.router)); +console.log(line("best single (dev)", heldOutSummary.bestSingle, bestSingle.seq)); +console.log(line("cheapest (dev)", heldOutSummary.cheapestSingle, cheapestSingle.seq)); +console.log(line("fixed cascade (dev)", heldOutSummary.bestFixedCascade, bestFixedCascade.seq)); +for (const b of baselines) { + const c = comparisons[b]; + const pp = (v) => `${(100 * v).toFixed(1)}`; + console.log( + ` router - ${b}: solve ${pp(c.solveRateDelta.estimate)} pp ` + + `[${pp(c.solveRateDelta.ci95[0])}, ${pp(c.solveRateDelta.ci95[1])}], ` + + `cost ${c.meanCostDelta.estimate.toFixed(3)} [${c.meanCostDelta.ci95[0].toFixed(3)}, ` + + `${c.meanCostDelta.ci95[1].toFixed(3)}] $/task; solved by only one: ` + + `router ${c.solvedByOnlyOne.router}, ${b} ${c.solvedByOnlyOne.baseline}`, + ); +} +console.log(`wrote ${outPath} in ${result.elapsedSeconds.toFixed(1)}s`); diff --git a/bench/universal-router/reproduce.sh b/bench/universal-router/reproduce.sh new file mode 100755 index 00000000..0480ec2b --- /dev/null +++ b/bench/universal-router/reproduce.sh @@ -0,0 +1,72 @@ +#!/bin/sh +# Reproduce data/router_prior.json from pinned public data, on a cold machine. +# +# sh bench/universal-router/reproduce.sh [WORK_DIR] +# +# Needs network access (Hugging Face, raw.githubusercontent.com, PyPI), python3 >= 3.10 with +# the venv module, and node >= 20. Everything goes under WORK_DIR (default: +# ${TMPDIR:-/tmp}/forgekit-router-repro), outside the repository: the downloaded per-task +# results are not redistributed here and must not be committed. +# +# 1. a venv with the pinned pyarrow (requirements.txt) +# 2. build_input.py: download + sha256-check the sources pinned in sources.json and write +# WORK_DIR/universal_all.json (a second run reads WORK_DIR/cache instead of downloading) +# 3. fit_prior.mjs: refit the prior into WORK_DIR/router_prior.json (several minutes of CPU). +# WORK_DIR/environment.txt records the versions, the CPU, the fit time, and which code the +# fit ran: the git HEAD and any uncommitted change to src/router, src/route.js or +# data/models.json. +# 4. compare_priors.mjs: compare it with data/router_prior.json; the report is +# WORK_DIR/prior-comparison.json +# +# Exit status: 0 when the refit reproduces every value of the shipped prior (except +# provenance.fittedAt; values only in the refit, such as provenance.build, are listed in the +# report as added); 1 when any value differs or is missing; anything else is an error in 1-3. +set -eu + +here=$(cd "$(dirname "$0")" && pwd) +root=$(cd "$here/../.." && pwd) +work=${1:-${TMPDIR:-/tmp}/forgekit-router-repro} +mkdir -p "$work" +work=$(cd "$work" && pwd) + +echo "== 1/4 python venv with the pinned pyarrow ($work/venv)" +[ -d "$work/venv" ] || python3 -m venv "$work/venv" +py="$work/venv/bin/python" +[ -x "$py" ] || py="$work/venv/Scripts/python.exe" +"$py" -m pip install --quiet --disable-pip-version-check -r "$here/requirements.txt" + +echo "== 2/4 build the fitting input" +"$py" "$here/build_input.py" --cache "$work/cache" --out "$work/universal_all.json" + +echo "== 3/4 refit the prior (this takes minutes; about 7 on a 2.8 GHz Xeon)" +# The code the fit loads, recorded just before it starts. +head=unknown +changes="unknown (not a git checkout)" +if git -C "$root" rev-parse --verify -q HEAD > /dev/null 2>&1; then + head=$(git -C "$root" rev-parse HEAD) + changes=$(git -C "$root" diff --stat HEAD -- src/router src/route.js data/models.json) + changes=${changes:-none} +fi +start=$(date +%s) +node "$here/fit_prior.mjs" "$work/universal_all.json" --out "$work/router_prior.json" +elapsed=$(($(date +%s) - start)) + +{ + echo "node: $(node --version)" + echo "python: $("$py" --version 2>&1)" + echo "pyarrow: $("$py" -c 'import pyarrow; print(pyarrow.__version__)')" + echo "cpu: $(node -p 'const c = require("os").cpus(); `${c[0]?.model} x${c.length}`')" + echo "platform: $(node -p 'process.platform + "-" + process.arch')" + echo "fit_elapsed_seconds: $elapsed" + echo "repo_head: $head" + echo "uncommitted changes to src/router, src/route.js, data/models.json:" + echo "$changes" +} > "$work/environment.txt" +cat "$work/environment.txt" + +echo "== 4/4 compare with data/router_prior.json" +status=0 +node "$here/compare_priors.mjs" "$root/data/router_prior.json" "$work/router_prior.json" \ + > "$work/prior-comparison.json" || status=$? +echo "report: $work/prior-comparison.json" +exit "$status" diff --git a/bench/universal-router/requirements.txt b/bench/universal-router/requirements.txt new file mode 100644 index 00000000..8a047b50 --- /dev/null +++ b/bench/universal-router/requirements.txt @@ -0,0 +1 @@ +pyarrow==25.0.1 diff --git a/bench/universal-router/sources.json b/bench/universal-router/sources.json new file mode 100644 index 00000000..11b6c86d --- /dev/null +++ b/bench/universal-router/sources.json @@ -0,0 +1,110 @@ +{ + "$comment": "Pinned inputs of bench/universal-router/build_input.py. Every file is checked against its sha256 before use; files from GitHub also carry their git blob id at the pinned commit. Keys of experiments.runs are registry ids (data/models.json), in canonical (ascending) order.", + "dataset": { + "name": "SWE-bench Verified", + "repo": "SWE-bench/SWE-bench_Verified", + "revision": "78f471bf655a3137b2e8a75af1501690ec009ec3", + "revisionDate": "2026-08-16", + "path": "data/test-00000-of-00001.parquet", + "url": "https://huggingface.co/datasets/SWE-bench/SWE-bench_Verified/resolve/78f471bf655a3137b2e8a75af1501690ec009ec3/data/test-00000-of-00001.parquet", + "bytes": 6304616, + "sha256": "030cfd7f2a704c4c0226e7f104c725a3b41230b1d3517f9c915ad7ea5be3fa25", + "note": "Revision 78f471b exists in the SWE-bench org repo, not in princeton-nlp/SWE-bench_Verified. The file is stored with Git LFS, so its sha256 is also its LFS oid." + }, + "experiments": { + "repo": "https://github.com/SWE-bench/experiments", + "commit": "40f164d5b8f1d249bf95a6df8b74b577fd8e519d", + "scaffold": "mini-SWE-agent 2.0.0", + "runDate": "2026-02-17", + "runs": { + "claude-haiku-4.5": { + "run": "20260217_mini-v2.0.0_claude-4-5-haiku-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_claude-4-5-haiku-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_claude-4-5-haiku-high/per_instance_details.json", + "gitBlob": "251ca4c02f45e8b109c0749821fa98414064daae", + "bytes": 53125, + "sha256": "4104e3cf041d57048ac22bf1ffc0805b154a064dfcbb1c0666647e4613008533" + }, + "claude-opus-4.5": { + "run": "20260217_mini-v2.0.0_claude-4-5-opus-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_claude-4-5-opus-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_claude-4-5-opus-high/per_instance_details.json", + "gitBlob": "39c730e7216071796617cb9c0a3c034b7c018d57", + "bytes": 52799, + "sha256": "e9657d90a145b9b22db233eb8a796d978ef9f8ecb9434570d0d599a635831907" + }, + "claude-opus-4.6": { + "run": "20260217_mini-v2.0.0_claude-4-6-opus", + "path": "evaluation/verified/20260217_mini-v2.0.0_claude-4-6-opus/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_claude-4-6-opus/per_instance_details.json", + "gitBlob": "082659fa7078a88eddb71c64b40c4f0e7f192ba2", + "bytes": 52805, + "sha256": "c669276c9b7c23ffe967df0d7cc9f1f6406001ae02095761ddb2ac067103430d" + }, + "claude-sonnet-4.5": { + "run": "20260217_mini-v2.0.0_claude-4-5-sonnet-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_claude-4-5-sonnet-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_claude-4-5-sonnet-high/per_instance_details.json", + "gitBlob": "697da24c32b43e81245ac92a741afa39f1b3293b", + "bytes": 52885, + "sha256": "c7f61049937deafa6327e0d5f17c0baf482751388530530bb9f5d53ce0fbf7cc" + }, + "deepseek-v3.2": { + "run": "20260217_mini-v2.0.0_deepseek-3-2-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_deepseek-3-2-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_deepseek-3-2-high/per_instance_details.json", + "gitBlob": "ad4a0b8333a866bb11d1849be15a809866fc8191", + "bytes": 53441, + "sha256": "5eeebd0aa53aa6234f36380d14141597f7776b32b0ba1a020dc2abaf005b50c1" + }, + "gemini-3-flash": { + "run": "20260217_mini-v2.0.0_gemini-3-flash-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_gemini-3-flash-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_gemini-3-flash-high/per_instance_details.json", + "gitBlob": "092530f4dc7eba8985a485aa1374af53436237d7", + "bytes": 53102, + "sha256": "6e49e555537839726c02dbbefd45cdf63bbed8c138eaf349d783381ca0716e33" + }, + "glm-5": { + "run": "20260217_mini-v2.0.0_glm-5-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_glm-5-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_glm-5-high/per_instance_details.json", + "gitBlob": "27bb601b3f59860985dcea4b682ad9a0e2feca5f", + "bytes": 53220, + "sha256": "cd139949a7a4044709f28d628c0a6ae9861da3e334033ae736a92a44d1d0a57c" + }, + "gpt-5-mini": { + "run": "20260217_mini-v2.0.0_gpt-5-mini", + "path": "evaluation/verified/20260217_mini-v2.0.0_gpt-5-mini/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_gpt-5-mini/per_instance_details.json", + "gitBlob": "22de38049e86540c4d93ddfa2ec6f21664360820", + "bytes": 52686, + "sha256": "b98b3f03cf9fbcf0eaa878ccde6d34a72483149ceb04559c90fd52277cb9d1ae" + }, + "gpt-5.2": { + "run": "20260217_mini-v2.0.0_gpt-5-2-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_gpt-5-2-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_gpt-5-2-high/per_instance_details.json", + "gitBlob": "cfdc6c07ad339e60e6d34b4d9050b85243b1bbc8", + "bytes": 52861, + "sha256": "8bac088fcc87374906a207a3cf93e25df715546ace895b3a588cc075f6353fd1" + }, + "kimi-k2.5": { + "run": "20260217_mini-v2.0.0_kimi-k2-5-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_kimi-k2-5-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_kimi-k2-5-high/per_instance_details.json", + "gitBlob": "d76d9549613a46e3ae98fcc727f8129cf9ae66b4", + "bytes": 53298, + "sha256": "3794b2c4b853845d97cf89db10f4264a0b27ba75f6b1bbbf1e34ddf6fa8a5c1f" + }, + "minimax-m2.5": { + "run": "20260217_mini-v2.0.0_minimax-2-5-high", + "path": "evaluation/verified/20260217_mini-v2.0.0_minimax-2-5-high/per_instance_details.json", + "url": "https://raw.githubusercontent.com/SWE-bench/experiments/40f164d5b8f1d249bf95a6df8b74b577fd8e519d/evaluation/verified/20260217_mini-v2.0.0_minimax-2-5-high/per_instance_details.json", + "gitBlob": "c14b9ea51db8dc65ff8fb9ae03366ddbe4906ce5", + "bytes": 53456, + "sha256": "62a26211006b557f1a31371ee6cf3119be5bdaed43dcce3fe6efedc0912c012c" + } + } + } +} diff --git a/docs/GUIDE.md b/docs/GUIDE.md index afd67525..8b298ff1 100644 --- a/docs/GUIDE.md +++ b/docs/GUIDE.md @@ -226,6 +226,24 @@ Run `forge route gateway` to emit a LiteLLM config so the routing happens automa Its tier aliases point at the same resolved ids, and each alias carries an `# id:` comment saying where its id came from (the live catalog, or the shipped snapshot and why). +`forge route universal ""` is a separate, opt-in router: it recommends a model or a +cascade across providers, minimising **expected** cost for the success you ask for +(`--objective match-best-single | target:p | value:V | budget:B`). Its shipped prior was fitted +on public SWE-bench Verified outcomes and refits exactly from pinned public data +(`bench/universal-router/reproduce.sh`); its held-out headline is repository-reported, from an +external harness, not independently reproduced, and a new in-repo held-out replay finds it no +better than a fixed cascade chosen on the same dev tasks. `budget:B` bounds expected cost, not what a task +can actually spend; an objective no cascade can meet is reported as infeasible (an explicit +`INFEASIBLE` line; `feasible: false` with the minimum achievable expected cost in `--json`). +A recommended model that no configured provider serves is marked `no provider id`, and the +recommendation is labelled `advice only` (`applicable: false`, `unmapped` in `--json`). +Being in the registry does not make a model callable. Add its id under `providers` in +`.forge/models.json`, or pass `--provider ` to route only among models you can call. +`forge route outcome` records one attempt's pass/fail, labelled self-reported unless +`--verify-run ` ties it to a `forge verify` run whose verdict agrees; `--attempt ` +makes recording idempotent. The model, its evidence status and its limits: +[docs/UNIVERSAL_ROUTING.md](UNIVERSAL_ROUTING.md). + ### `forge models` — what each tier resolves to Forge ships no model id it depends on. `src/model_tiers.json` names each tier's **family** @@ -658,13 +676,49 @@ $ forge verify Forge verify changed files: 2 - tests: ✓ pass + tests: ✓ pass (npm test, npm test (packages/api)) + suites ran: npm test=PASS, npm test (packages/api)=PASS + packages: 2/2 covered symbols checked: 7 - provenance: .forge/provenance.json + provenance: .forge/provenance.json (run d83f0b7f-45f1-466c-8e3b-9754b26cf356) PASS ``` +**What a PASS covers.** Every package that declares its own suite is planned and run in its +own directory: the root, plus each nested package with an explicit `scripts.test`, a pytest +config, a `go.mod`, and so on. `packages: n/m covered` counts the packages that reached a +verdict. If any package's suite never reaches one, the result is `INCOMPLETE`, never `PASS`. +If any package fails, the result is `FAIL`, however green the root is. A root script that +already runs every workspace (`npm test --workspaces`, `pnpm -r test`, `yarn workspaces +foreach`, `turbo run test`, `lerna run test`, `nx run-many`) covers them in one run instead +of once per package. Fixture and test-data packages are never required. A test runner that is +only a devDependency is not an obligation either; `forge stack` lists it as `available`. +Tune this per repo under `verify` in `.forge/forge.config.json`: + +```json +{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"] } } +``` + +- `workspaces: "root"` declares that the root command already covers every package. +- `exclude` lists package paths that are not required suites. +- `generated` lists outputs a test run may legitimately write. + +**Bound to the code that was tested.** The stamp is bound to a fingerprint of the working +tree: HEAD, the staged and unstaged diffs, and each untracked file's path, mode, size and +content hash. The fingerprint is taken before AND after the run. If something changed the code +while the tests ran (a formatter, a code generator, another agent), the result is `INCOMPLETE` +with `mutated: true`, and the stamp names the pre-run state. Interpreter caches +(`__pycache__`, `.pytest_cache`, …) and the `generated` paths never count as a change. An +untracked file that cannot be read makes the state unbindable; it is never silently skipped. +Stamps written before this fingerprint (scheme `manifest-v2`) no longer verify: re-run +`forge verify`. + +Every run also appends one event to `.forge/verify-events.jsonl`: the run id, the verifier +and its version, the suites, the coverage, the pre- and post-run state, and an environment +digest, sealed with a machine-local MAC. The event is never rewritten. `forge route outcome +--verify-run ` ties a routing outcome to it. + **`forge verify --deep` — multi-lens consensus.** The deep mode runs a table of independent lenses over the same diff — the test suite, unknown symbols, atlas dependents the diff never touched, code-without-docs drift, secret-shaped tokens in @@ -739,10 +793,16 @@ $ forge stack languages: JavaScript/TypeScript, TypeScript frameworks: Next.js, React pkg mgrs: pnpm - test: npx vitest + test: pnpm test + available: npx vitest evidence: package.json ``` +`test` is what the repo declares it runs: an explicit `scripts.test` wins, and npm's +`"no test specified"` placeholder is not a suite. `available` lists runners that are +installed (a devDependency) but not what the repo runs. They are inventory, not an +obligation: `forge verify` never requires them. + Detection reads `package.json` (deps → frameworks, lockfile → package manager), `pyproject.toml`/`requirements.txt`, `go.mod`, `Cargo.toml`, `Gemfile`, `composer.json`, `pom.xml`/`build.gradle`, and `*.csproj`. Widening it is adding a data row, not code. @@ -919,7 +979,8 @@ before canonicalization folded CRLF into LF: those carry their pre-fold id in th which reads still accept, so nothing is broken without it — but the old form and a teammate's freshly minted copy of the same fact stay two entries until you run it. It moves each claim's evidence and provenance logs with it, unions them into an existing twin rather than -overwriting, and is idempotent. +overwriting, and is idempotent. Add `--dry-run` to preview the migration: it prints what +would move and what would merge, and writes nothing. `forge ledger compact [--dry-run]` archives what this ledger's own history says will not be used again. It prints every number it learned, and nothing in it is a fixed threshold: @@ -947,8 +1008,18 @@ Forge ledger — compact (every cut-off learned from this ledger) [dry run] **Where use comes from:** forge writes `.forge/ledger/.usage.jsonl`, a gitignored local log. It records each claim that the session lesson block, pre-edit lessons, the déjà-vu advisory, `ledger query` or the MCP query served. +**Similar is not the same.** Two claims are grouped as duplicates only when they also agree +on everything that changes behaviour: operators, numbers and units, quoted literals, +identifiers, paths, and negation. "Enable authentication…" and "Disable authentication…" are +never merged, however similar their words. Such a pair is printed under `kept apart — similar +but conflicting`, so a human can retract the wrong one. + **What happens to archived claims:** - They move to `.forge/ledger/attic/`, and their logs stay where they are. +- Each one records WHY it was archived in `attic/.log`: `tombstoned`, `dormant`, `idle`, + or `duplicate` (naming the claim that was kept). Archived is not refuted: consolidation + drops a learned lesson only when its claim was actually retracted or went dormant. An idle + archive keeps the lesson, and a duplicate defers to the claim that was kept. - `forge ledger show` and `blame` still read them. - New evidence brings one back. - The Stop hook applies the first two rules on its own; duplicates are grouped only by this command. @@ -1026,9 +1097,22 @@ tooling migrates. ### `forge reuse` — proof-carrying code cache -Verified code becomes an `artifact` claim keyed by a normalized task fingerprint; a -lookup walks exact → near → adapt → miss. An artifact serves **only while its proof -holds** — confidence above the 0.6 floor and every declared dependency still in the atlas. +Verified code becomes an `artifact` claim keyed by its task text; a lookup walks exact → +near → adapt → miss. An artifact serves **only while its proof holds**: confidence above the +0.6 floor, its file unchanged since it was minted, and every declared dependency still in +the atlas with the same declaration. + +- **Exact means the same text.** The key ignores only whitespace (and Unicode + normalization). Case, operators, literals and punctuation all count, so `age >= 18` and + `age <= 18`, or `"ADMIN"` and `"admin"`, never share a key. Artifacts minted before this + key was introduced never hit exact. +- **Near must also agree on behaviour.** A reworded match is offered as `near` only when + the two specs share their operators, numbers, literals, identifiers, paths and negation. + Otherwise it drops to `adapt`, with a note naming what differs. +- **Checked where it is served.** An artifact whose file was edited or deleted since it was + minted is not served. When a dependency's declaration changed, it is not served either. + With no atlas to check against, the hit is marked `NOT revalidated` (`requiresRevalidation: + true` in `--json`), never presented as checked. ```console $ forge reuse query "debounce user input before firing search" @@ -1090,7 +1174,7 @@ difference — not a feeling. $ forge context "change verifyToken in src/auth.js to reject short tokens" Forge context — budgeted assembly + completeness gate - budget: 1840/12000 tokens · required 4 · COMPLETE + budget: 1840/12000 tokens (chars/3.6 estimate of the rendered block) · required 4 · COMPLETE + def:src/auth.js [full] 620t + deps:verifyToken [head] 410t + tests:test/auth.test.js [full] 480t @@ -1100,7 +1184,27 @@ Forge context — budgeted assembly + completeness gate On an incomplete assembly it lists the missing items and derived clarifying questions ("the task names `X` but the repo doesn't define it — which file implements it?") and exits 1. `--budget ` tightens the window; a tight budget downgrades granularity -instead of silently dropping coverage. +instead of silently dropping coverage, and reports what it could not deliver. + +What `COMPLETE` (`ok: true`) does and does not mean: + +- **Syntactically delivered, not semantically sufficient.** Every required item was delivered + as content inside the budget. Whether that content is enough for the edit is not measured. +- **Token counts are an estimate** — characters ÷ 3.6 of the rendered block, labels and + separators included — not a model tokenizer's count. +- **A pointer is not coverage.** An item that only fits as a one-line `- read ` pointer + is a pending read obligation, listed under `pending`, and the assembly is not complete until it + is read. A 25-line head covers a definition only when the definition's line is inside it, and + a dependents list cut at 12 names what it left out (`truncated`). +- **Over budget is INCOMPLETE.** When even pointers do not fit, the result says `overflow: true` + and `ok: false`; it never reports a silent over-budget pass. +- **Selection is a heuristic.** Optional items are picked greedily by value density (score ÷ + tokens) with per-source diminishing returns, with no optimality guarantee. +- **Only the explicit paths assemble.** `forge substrate` and `forge context` build the + selection; the ambient per-prompt hook does not (latency), it works from caches. + +Status and contract: [substrate-v2 plan 04 §7](plans/substrate-v2/04-context-assembly.md#7-status-2026-09-26--partial) +(P4 is partial). ### `forge diagnose ""` — doom-loop check @@ -1149,13 +1253,19 @@ Forge imagine — consequence simulation (pre-action) minimal dry-run suite (1) — run these, in this order: - test/auth.test.js - (measure it: re-run with --run — sandboxed worktree dry-run of HEAD) + (measure it: re-run with --run — a dry-run in an isolated checkout of HEAD; not a security sandbox) ``` -Add **`--run`** to actually execute that suite in a sandboxed worktree of HEAD — the -dry-run result lands as oracle evidence on the prediction. It refuses a dirty working -tree (your uncommitted changes wouldn't be in the run); commit/stash first or pass -`--allow-dirty` to knowingly measure the last commit. It also flags predicted breaks +Add **`--run`** to actually execute that suite in an isolated git checkout (a detached-HEAD +worktree) — the dry-run result lands as oracle evidence on the prediction. The checkout isolates +files only: it is not a security sandbox, and the tests run with your network, credentials, home +directory and process permissions. It tests the committed baseline, not a proposed uncommitted +patch, so it refuses a dirty working tree (your uncommitted changes wouldn't be in the run); +commit/stash first or pass `--allow-dirty` to knowingly measure the last commit. The selected +files run under `node --test` and nothing else: when a selected test is not a `.js`, `.mjs` or +`.cjs` file, or the project's `test` script uses another runner (Jest, Vitest, Mocha, pytest), +the result is an explicit **unsupported runner** verdict instead of a run — run the project's own +runner on the selected files. It also flags predicted breaks **no test covers** — the risk you can't dry-run away. ### `forge uicheck` — deterministic UI checks @@ -1163,6 +1273,15 @@ tree (your uncommitted changes wouldn't be in the run); commit/stash first or pa Five subcommands: three are static parsing — no LLM, no screenshots — and `visual` and `interact` optionally drive a real browser. +They are **advisory measurements of different things**: contrast arithmetic, design-token and +scale conformance, distance from generic templates, fingerprint similarity, and four rendered +behaviours. A passing contrast check is not an accessibility result, and distance from common +templates is not user value. `contrast`, `design` and `visual` exit 1 on a fail so a script or CI +step can gate on them if you choose; nothing runs them as a blocking hook, and `interact` only +gates under `--enforce`. What each measures, and the labelled set, false-positive cost and +exception path a check would need before becoming blocking: +[substrate-v2 plan 07 §7](plans/substrate-v2/07-ui-quality-gate.md#7-ui-checks-are-advisory--what-each-one-measures-2026-09-26). + **`contrast [--large] [--json]`** — exact WCAG math, asserted, never guessed (bare `forge uicheck ` still works). It **exits 1 when the pair fails AA**, so a script or CI step can gate on it. `--large` grades against the large-text / UI-component @@ -1308,16 +1427,20 @@ Forge uicheck interact — browser interaction checks A read-only lens over `.forge/` — stdlib `node:http`, localhost-only, one self-contained HTML page (no CDN, no build step). Panels: Ledger (claims with val bars, -contested claims, per-author trust), Cost/Cache (measured stage counters), Impact +contested claims, per-author trust), Cost/Cache (measured stage counters — stage +self-estimates, not end-to-end spend; a stage with no logs is unknown, not $0), Impact (blast-radius explorer), Radar (dependency-currency rings read from the `.forge/radar.json` cache), Trends (per-stage metrics history as inline-SVG sparklines), Memory browser (ranked recall search over the ledger with confidence + freshness bars), and Session timeline (durable mint/tombstone events across sessions). The page live-refreshes every 5s, paused while the tab is hidden. Every claim row shows its `forge ledger blame` command. -The only two writes are the human-driven `POST /api/ratify` and `POST /api/retract`; both -are guarded against CSRF and DNS-rebinding — a request whose `Host` isn't the loopback -interface, or whose browser `Origin` isn't the loopback origin, is refused with `403` (a -non-loopback `--host` bind is the documented opt-out). +The only two writes are the human-driven `POST /api/ratify` and `POST /api/retract`. +Every route, reads included, refuses a request whose `Host` is not this loopback address +and port, with `403`; that blocks DNS rebinding. A write also needs the per-session token +the server embeds in its own page (sent back as the `x-forge-token` header). When a browser +sends an `Origin`, it must be exactly this server's origin, so a page on another localhost +port cannot write either. A script that POSTs must read the token from the page first. +Binding a non-loopback `--host` turns the `Host` check off; that exposure is your choice. ```console $ forge dash @@ -1370,10 +1493,23 @@ Read the `context` line with care: that 62% is the white paper's 30-task routing demonstration, measured on the tasks its thresholds were tuned on. A pre-registered evaluation on 80 held-out tasks refuted it — counting every escalation, routing cost 20.2% _more_ than always-premium ([research/empirical-refutation/](../research/empirical-refutation/)). -The line is quoted here as the CLI currently prints it. +The line is quoted here as the CLI currently prints it. The ~90 % "target" it names is a +hypothesis, never a result: the stage arithmetic it was argued from used the refuted routing +figure and has been withdrawn (see the plan's cost model, §2). Plain `forge cost` remains the per-day spend view via `ccusage`. +**Reading any cost number.** A figure is only comparable when it states: the currency and the +date of the prices; whether it is actual spend or a counterfactual (tokens repriced at another +model's price are a counterfactual, not an observed saving); which attempts it includes (first +attempt only, or every retry and escalation); whether cached tokens and verifier and tool costs are +counted; and what data is missing. A day or stage with no logs is **unknown, not $0 actual +spend**. The number that decides whether the substrate saves money is total cost per completed, +externally verified task, read beside the acceptance and abandonment rates, against a baseline +with equivalent tools, context and repair opportunity — always-premium / read-everything is a weak +comparator. Per-stage factors diagnose where cost goes; they are not multiplied into a total. The +full rule: [substrate-v2 plan 05 §3](plans/substrate-v2/05-cost-model.md#3-acceptance-rule-for-any-cost-headline). + ### The rest | Command | Answers | @@ -1689,9 +1825,10 @@ session. For learned-from-mistakes memory, just work: Cortex captures recurring corrections on its own. **UI work.** `forge uicheck contrast ` for exact contrast; `forge uicheck -design --taste -
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory, foresight, and guardrails—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
886ms
End-to-end benchmark
Blast-radius scan
0.40ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
+
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory and foresight—with automatic guardrails on Claude Code—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
851ms
End-to-end benchmark
Blast-radius scan
1.68ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
  /plugin install forgekit
 

Recommended · ambient guards on every prompt

diff --git a/mintlify/cli/substrate.mdx b/mintlify/cli/substrate.mdx index f399e3d9..eb8439b0 100644 --- a/mintlify/cli/substrate.mdx +++ b/mintlify/cli/substrate.mdx @@ -110,9 +110,13 @@ Consequence simulation — predicted breaks + the minimal dry-run test suite for ```bash forge imagine "" -forge imagine "" --run # execute the minimal suite sandboxed +forge imagine "" --run # execute the minimal suite in an isolated git checkout ``` +`--run` executes the suite in an isolated git checkout (a detached-HEAD worktree). It isolates +checkout files only — not network, credentials, your home directory or process permissions — and +it tests the committed baseline, not uncommitted changes. + ## `forge lean` Scope-minimality (M5) — measure the diff's footprint vs what the task asked for. diff --git a/mintlify/introduction.mdx b/mintlify/introduction.mdx index 144c5307..8808cc9c 100644 --- a/mintlify/introduction.mdx +++ b/mintlify/introduction.mdx @@ -1,17 +1,21 @@ --- title: "Forge: the cognitive substrate for AI coding agents" -description: "Forge is the cognitive substrate stateless models are missing — memory, foresight, and guardrails — as native config for every AI coding agent." +description: "A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, and explicit verification around coding agents — as native config for every AI coding tool." --- -**One brain for every AI coding agent.** A large language model is stateless: one -context window, wiped every call. It has no memory of what your team learned, no -foresight about what an edit will break, and no enforced guardrails. Forge +**A beta toolkit for shared evidence-referenced memory, heuristic change-impact analysis, +and explicit verification around coding agents.** A language model keeps no durable state between +independent calls and sees only a bounded context window, so on its own it does not carry what +your team learned, cannot see the parts of the repository an edit affects unless they are in +context, and cannot enforce rules on itself. Forge (`@codewithjuber/forgekit`) is the **cognitive substrate** — the layer that runs _before_ the model edits code, supplying evidence-referenced, content-addressed memory (we -call it "proof-carrying memory"), heuristic impact foresight, and enforced guardrails — and -a **cross-tool config compiler** that delivers that brain as native config into every tool -at once. Claude Code is the deepest-tested integration; the others receive native config and -MCP tools with less real-world exercise. +call it "proof-carrying memory"), heuristic impact foresight, and guardrails (blocking on +Claude Code) — and a **cross-tool config compiler** that delivers that brain as native config +into every tool at once. Claude Code is the deepest-tested integration and the only one with +automatic hooks; the others receive native config (most also an MCP server entry) with less +real-world exercise. The repository's `docs/INTEGRATIONS.md` lists, per tool, what is emitted, +registered, run automatically and enforced. @@ -30,10 +34,11 @@ MCP tools with less real-world exercise. ## The problem -A large language model is stateless — one context window, wiped every call. +A language model keeps no durable state between independent calls, and it sees only what fits in +its context window. -- It has **no memory** of what your team already learned. -- It has **no foresight** about what an edit will break. +- It does not **carry what your team learned** from one session to the next. +- It does not **see what an edit will affect** unless those files are in its context. - It has **no enforced guardrails** — prose rules get forgotten after a compaction. And every tool wants its own config file (`CLAUDE.md`, `AGENTS.md`, `.cursor/rules`, @@ -42,11 +47,12 @@ things, and the compiler that delivers it into every tool from one source. ## The thesis -A model can't learn from your codebase between calls: its weights are frozen and its -working memory is wiped after every response. Memory, foresight, and self-checking -can't be prompted into it — they have to be supplied from _outside_. That outside layer -is the cognitive substrate. Formally, inference is a fixed function `y = f(x)` with no -state between calls; Forge is the state. +A model with frozen weights does adapt within one context — examples, retrieved facts and +feedback change what it does — but it does not guarantee four things on its own: durable state +across independent calls, context beyond its window, weight updates from outcomes, and reliable +self-verification without external evidence. Forge supplies persistence and external checks from +_outside_ the model. It is one tested way of doing that, not the only possible architecture; +the research programme's dated corrections explain the difference. @@ -75,8 +81,9 @@ state between calls; Forge is the state. content-addressed memory: a claim that carries references to its evidence and is trusted only once independent oracles raise its confidence above a floor. The "proof" is that evidence trail, not a formal proof. -- **Foresight before you break things.** Ask "what does changing `verifyToken` break?" - and get the blast radius from the code graph, including coupled files you never named. +- **Heuristic impact before you break things.** Ask "what does changing `verifyToken` + break?" and get the blast radius from a regex-derived code graph, including coupled files you + never named — it can miss files as well as over-warn. - **Guardrails that can't be forgotten.** Deterministic hooks enforce protected paths, cost budgets, and doom-loop detection — they survive a context compaction. - **Work that finishes end to end.** A completion gate blocks "done" once per session @@ -117,8 +124,10 @@ Forge states its own ceiling everywhere. - **Tests and human corrections always win.** - Forge is **beta**. The core (`init`, `sync`, `substrate`, `impact`, `ledger`, guards) - is tested and in daily use; some flags may change before `1.0`. + Forge is **beta**: releases follow semantic versioning (a breaking change is a new major + version), and "beta" describes maturity — heuristic analyses, advisory checks, and less + real-world exercise outside Claude Code. The core (`init`, `sync`, `substrate`, `impact`, + `ledger`, guards) is tested and in daily use. ## Next steps diff --git a/reports/2026-07-05.md b/reports/2026-07-05.md index f4542af8..8e89e492 100644 --- a/reports/2026-07-05.md +++ b/reports/2026-07-05.md @@ -1,5 +1,11 @@ # Claude Config Radar — 2026-07-05 +> ⚠️ **Archived — historical.** An auto-generated daily report from 2026-07-05, kept for the record. +> Not maintained and not the current state; its action items (including the credential-rotation +> reminders, which mention no secret values) are historical and superseded. See +> **[docs/GUIDE.md](../docs/GUIDE.md)** for today's tooling and **[SECURITY.md](../SECURITY.md)** for +> reporting a live security issue. + **TL;DR** - 🔴 **Security-relevant dep bump**: pgvector **0.8.2** fixes a buffer overflow in parallel HNSW index builds (**CVE-2026-3172**) — your stack uses pgvector, so this is the one worth acting on. - 🟡 **shadcn/ui** made **Base UI the default** component library (docs + new projects) as of July 2026 — escalation of yesterday's "picking Base UI ~2:1" note; Radix not deprecated. diff --git a/reports/benchmarks.md b/reports/benchmarks.md index b52e2f9c..e27ae4c1 100644 --- a/reports/benchmarks.md +++ b/reports/benchmarks.md @@ -4,7 +4,7 @@ > **a number is an assumption until measured.** Every figure in the generated section below > came from an actual run of `npm run bench` on the machine recorded in the environment > block — no projections, no targets, no numbers copied forward from a different machine. -> Re-run `npm run bench` (≈10 s, node stdlib only) and the generated section is rewritten +> Re-run `npm run bench` (≈20 s, node stdlib only) and the generated section is rewritten > in place with your machine's numbers. ## Methodology @@ -45,12 +45,22 @@ - **ledger / val()**: pure in-memory scoring; the fixture gives most claims 0–1 evidence records, and val() cost scales with evidence count — a heavily-evidenced ledger will be slower per claim. -- **reuse / lookup**: the memoized `_sketch` cache is stripped before every timed run, so - each run behaves like a fresh CLI process. The *exact* tier returns before any pool - sketching (normalized-string compare); the *near* tier pays MinHash-sketching the whole - candidate pool plus LSH banding — that difference is the point of reporting both. +- **reuse / lookup**: every row is VALIDATED before it is timed — the fixture's artifacts + cite a real, resolvable git object as their test evidence, and the harness aborts unless + the lookup returns the tier the row is labeled with (review F14, 2026-09-26: the previous + fixture cited untyped `bench:artifact:` refs, which val() caps below the serving floor, + so every "exact"/"near" row had actually measured a MISS; those older numbers are + invalid). "cold" rows strip every memoized sketch the lookup path caches (`_sketch`, + `_terms`, `_specSketch`, `_keySketch` — the old harness stripped only `_sketch`, which the + reuse ladder never reads), so each run behaves like a fresh CLI process; "warm" rows keep + them, like a long-lived process. The *exact* tier returns before any pool sketching + (identity-key compare); *near* and *miss* pay MinHash-sketching the candidate pool plus + LSH banding — that difference is the point of reporting them separately. - **context / assemble()**: warm atlas, empty ledger (the repo copy has no `.forge`), includes the real file reads for pinned items. Task: a three-symbol, one-file edit spec. + "complete"/"incomplete" is the assembler's own honest verdict: an item that could only + be delivered as a pointer or partial span is a pending read, so a large named file makes + this task's context incomplete at the default budget (review F02/F03). - **substrate / substrateCheck**: the whole deterministic gate — preflight grounding, routing rubric, up to 8 impact queries, reuse lookup, context assembly, scope decomposition, lessons, minimality, goal anchor — with `llm: false`. **No model latency @@ -90,9 +100,12 @@ now resolve to the exact symbol (`src/atlas.js:17 imports → src/util.js:conten What these numbers do **not** mean: n = 6 cases, one JavaScript repo, symbols chosen to be uniquely named (the atlas resolves ambiguous names to nothing — a separate, known -limitation). They are not comparable to the paper's numbers, which came from mutation -testing a Python codebase against a real test suite. The two appear side by side below, -labeled, and are never blended. +limitation). They are not comparable to the paper prototype's numbers: its 0.63 / 1.00 / +0.75 came from mutation testing on the authors' own fixture — a self-built demo that the +pre-registered field study REFUTED (pooled precision 0.40, recall 0.022, F1 0.042 over 759 +files' mined co-change in nine repositories; see `research/empirical-refutation/`). The +regex atlas here is a different, Node graph — not the evaluated Python oracle. All three +appear side by side below, labeled, and are never blended. > **History of this row.** The precision 0.90 / F1 0.92 this file carried until 2026-09-21 came > from a much smaller atlas (145 files) and a reverse walk that stopped at the direct @@ -114,61 +127,69 @@ labeled, and are never blended. ```json { - "node": "v24.19.0", - "cpu": "AMD EPYC Processor (with IBPB)", + "node": "v22.22.2", + "cpu": "Intel(R) Xeon(R) Processor @ 2.80GHz", "cores": 4, - "memGB": 8, - "platform": "win32", + "memGB": 16, + "platform": "linux", "arch": "x64", - "commit": "703da31d574c30d22bef019b1c8563ade0d0d6be", - "date": "2026-09-21T22:09:58.366Z" + "fsType": "ext2/ext3", + "commit": "a56606afd5baebdabee95ad95af626a965557d92 + uncommitted changes", + "date": "2026-09-26T20:35:15.125Z" } ``` ### Measured results -| suite | benchmark | median | p95 | runs | notes | -|-----------|---------------------------------------------|----------|---------|------|----------------------------------------| -| atlas | full build (this repo) | 530 ms | 622 ms | 5 | 455 files, 10498 symbols, 29728 edges | -| atlas | incremental rebuild (unchanged) | 339 ms | 359 ms | 5 | per-file hash cache hit | -| atlas | impact("claimText") (warm adjacency) | 0.40 ms | 1.08 ms | 30 | 51 files impacted | -| ledger | mint+put 1000 claims | 1854 ms | 1986 ms | 5 | 539/s | -| ledger | loadClaims at 1000 claims | 213 ms | 230 ms | 5 | full state from disk | -| ledger | mergeDirs 2×500-claim replicas (250 shared) | 4308 ms | 4409 ms | 3 | +250 claims, +313 records | -| ledger | val() over 1000 claims | 0.076 ms | 0.16 ms | 20 | 13,140,604/s (mean val 0.51) | -| reuse | fingerprint 2000 specs | 116 ms | 156 ms | 5 | 17,171/s | -| reuse | lookup exact @ 100 artifacts | 5.71 ms | 10.5 ms | 10 | tier=miss | -| reuse | lookup near (LSH) @ 100 artifacts | 4.76 ms | 5.18 ms | 5 | tier=miss, j=- | -| reuse | lookup exact @ 1000 artifacts | 52.8 ms | 88.5 ms | 10 | tier=miss | -| reuse | lookup near (LSH) @ 1000 artifacts | 46.7 ms | 89.8 ms | 5 | tier=miss, j=- | -| context | assemble() (this repo, 3-symbol task) | 12.6 ms | 30.0 ms | 10 | 4070/6000 tokens, 9 required, complete | -| substrate | substrateCheck (allowBuild, llm off) | 886 ms | 908 ms | 3 | 99 impacted files, route simple | +| suite | benchmark | median | p95 | runs | notes | +|-----------|----------------------------------------------|---------|---------|------|--------------------------------------------------------------| +| atlas | full build (this repo) | 719 ms | 753 ms | 5 | 504 files, 13396 symbols, 37369 edges, cap 20000 not reached | +| atlas | incremental rebuild (unchanged) | 346 ms | 365 ms | 5 | per-file hash cache hit | +| atlas | impact("claimText") (warm adjacency) | 1.68 ms | 2.38 ms | 30 | 61 files impacted | +| ledger | mint+put 1000 claims | 248 ms | 326 ms | 5 | 4,038/s | +| ledger | loadClaims at 1000 claims | 13.6 ms | 15.3 ms | 5 | full state from disk | +| ledger | mergeDirs 2×500-claim replicas (250 shared) | 191 ms | 198 ms | 3 | +250 claims, +313 records | +| ledger | val() over 1000 claims | 0.85 ms | 1.39 ms | 20 | 1,170,474/s (mean val 0.51) | +| reuse | fingerprint 2000 specs | 232 ms | 290 ms | 5 | 8,636/s | +| reuse | lookup exact hit, cold @ 100 artifacts | 1.01 ms | 9.43 ms | 10 | tier=exact | +| reuse | lookup exact hit, warm @ 100 artifacts | 0.67 ms | 5.89 ms | 10 | tier=exact | +| reuse | lookup near hit (LSH), cold @ 100 artifacts | 18.5 ms | 33.0 ms | 5 | tier=near, j=0.98 | +| reuse | lookup miss, cold @ 100 artifacts | 18.2 ms | 25.4 ms | 5 | tier=miss | +| reuse | lookup exact hit, cold @ 1000 artifacts | 3.98 ms | 6.11 ms | 10 | tier=exact | +| reuse | lookup exact hit, warm @ 1000 artifacts | 3.38 ms | 4.14 ms | 10 | tier=exact | +| reuse | lookup near hit (LSH), cold @ 1000 artifacts | 128 ms | 133 ms | 5 | tier=near, j=0.95 | +| reuse | lookup miss, cold @ 1000 artifacts | 123 ms | 130 ms | 5 | tier=miss | +| context | assemble() (this repo, 3-symbol task) | 17.6 ms | 27.0 ms | 10 | 4174/6000 tokens, 9 required, incomplete | +| substrate | substrateCheck (allowBuild, llm off) | 851 ms | 995 ms | 3 | 141 impacted files, route simple | ### Impact-oracle quality (hand-labeled cases, this repo) | case (target) | precision | recall | F1 | predicted | truth | |---------------|-----------|--------|------|-----------|-------| -| normalizeSpec | 0.12 | 1.00 | 0.21 | 17 | 2 | +| normalizeSpec | 0.11 | 1.00 | 0.20 | 18 | 2 | | evalImpact | 0.29 | 1.00 | 0.44 | 7 | 2 | -| isStale | 0.20 | 1.00 | 0.33 | 30 | 6 | -| mergeStates | 0.16 | 1.00 | 0.28 | 25 | 4 | -| claimText | 0.16 | 1.00 | 0.27 | 51 | 8 | -| contentHash | 0.11 | 1.00 | 0.21 | 87 | 10 | -| mean of 6 | 0.17 | 1.00 | 0.29 | | | +| isStale | 0.17 | 1.00 | 0.30 | 40 | 7 | +| mergeStates | 0.14 | 1.00 | 0.25 | 28 | 4 | +| claimText | 0.18 | 1.00 | 0.31 | 61 | 11 | +| contentHash | 0.10 | 1.00 | 0.19 | 105 | 11 | +| mean of 6 | 0.17 | 1.00 | 0.28 | | | -Edited-file-only baseline recall over the same cases: **0.27**. +Edited-file-only baseline recall over the same cases: **0.26**. -Two methodologies, side by side — different codebases, different ground-truth +Different methodologies, side by side — different codebases, different ground-truth derivations, so the rows are comparable in spirit only and are never blended: -| series | precision | recall | F1 | ground truth | -|--------------------------------------------|-----------|--------|------|-----------------------------------------------| -| paper prototype (Python, mutation-derived) | 0.63 | 1.00 | 0.75 | mutation testing against a real suite | -| this repo (regex atlas, hand-labeled) | 0.17 | 1.00 | 0.29 | 6 hand-labeled cases (bench/impact_cases.mjs) | +| series | precision | recall | F1 | ground truth | +|------------------------------------------------|-----------|--------|------|------------------------------------------------------------| +| paper prototype, self-built demo (REFUTED) | 0.63 | 1.00 | 0.75 | mutation testing on the authors' own fixture | +| paper prototype, field study (pooled, 9 repos) | 0.40 | 0.02 | 0.04 | 759 files' mined co-change (research/empirical-refutation) | +| this repo (regex atlas, hand-labeled) | 0.17 | 1.00 | 0.28 | 6 hand-labeled cases (bench/impact_cases.mjs) | -> **Snapshot boundary.** The measured results above were generated at commit `eb68ea9` and +> **Snapshot boundary.** The measured results above were generated at the commit recorded in +> the environment block (`commit`; "+ uncommitted changes" means the working tree the review +> fixes of 2026-09-26 were measured in, before they were committed) and > do not benchmark the optional embedding adapter now implemented in `src/embed.js` and > exercised with a deterministic fake provider in `test/embed.test.js`. MinHash remains the > zero-dependency default and failure fallback. The structural comparisons below describe the @@ -216,6 +237,6 @@ that forgekit structurally does not. ## Reproduce ```sh -npm run bench # ≈10 s; prints the tables and rewrites the generated section above +npm run bench # ≈20 s; prints the tables and rewrites the generated section above npm test # includes a smoke test of the harness's pure helpers (test/bench.test.js) ``` diff --git a/reports/cost-eval.md b/reports/cost-eval.md index 0c66fd25..7171f28f 100644 --- a/reports/cost-eval.md +++ b/reports/cost-eval.md @@ -6,21 +6,30 @@ > saving. The paper's 62 % routing saving (paper §9) was measured on the 30 tasks its > thresholds were tuned on and is **refuted**: on 80 held-out tasks, counting every escalation, > routing cost 20.2 % _more_ than always-premium ([research/empirical-refutation/](../research/empirical-refutation/)). -> The plan's ~90 % composed figure is a **target**, not a result, and does not appear in this table. +> The plan's ~90 % figure is a **target** and a hypothesis, not a result, and does not appear in this table. +> +> **Corrected 2026-09-26.** The methodology below used to say the cost model "is multiplicative — +> `C = C₀ · Π(1 − fᵢ)` over independent stages — so each stage factor is measured separately and +> composed arithmetically", and that paired runs reprice "identical tokens". Stage savings interact +> (cache hits change the routed workload, context changes retries, halts can defer work), so the +> per-stage factors are diagnostics, not a total; and repricing tokens at another model's price is a +> counterfactual, not an observed outcome. The acceptance rule for any cost headline is in +> [05-cost-model.md §3](../docs/plans/substrate-v2/05-cost-model.md#3-acceptance-rule-for-any-cost-headline). ## Methodology -The cost model is multiplicative — `C = C₀ · Π(1 − fᵢ)` over independent stages — so each -stage factor is measured separately and composed arithmetically, never asserted: +Each stage factor is measured separately as a diagnostic; the system is judged only on paired, +full-system outcomes (total cost per completed, externally verified task), never on a product of +stage factors: 1. **Instrumentation.** Every substrate stage appends one line to `.forge/metrics.jsonl` (`{t, stage, outcome, tokensIn, tokensOut, tier, savedEstimate, ref}` — `src/metrics.js`). `forge cost --stages` computes the per-stage factors from those lines (`src/cost_report.js`); a stage with no events reports **no data**, never a default. -2. **Paired runs.** Baseline (always-premium, read-everything, no cache) vs. substrate over - the same replay corpus (N ≥ 100 real tasks, stratified repeat-heavy / mixed / cold), the - paper §9 methodology: identical tokens repriced, so every saving is arithmetic on measured - tokens. +2. **Paired runs.** Baseline (equivalent tools, context and repair opportunity; always-premium / + read-everything reported too) vs. substrate over the same replay corpus (N ≥ 100 real tasks, + stratified repeat-heavy / mixed / cold), each policy actually executed. Repriced tokens are a + labelled counterfactual, not a measured saving. 3. **Correctness guard (spec §3).** A saving counts only if the external verifier passes the output. A routed-down answer that fails is not a saving; a cache hit that gets reverted is recorded as a *negative* entry. @@ -33,7 +42,7 @@ stage factor is measured separately and composed arithmetically, never asserted: | cache (reuse, tier-weighted) | — | 0 | no data yet — run with metrics enabled | | route (vs always-premium) | — | 0 | no data yet — run with metrics enabled | | context (assembly ρ) | — | 0 | no data yet — run with metrics enabled | -| **composed (measured stages only)** | — | 0 | nothing to compose yet | +| **composed (measured stages only; diagnostic, not a total)** | — | 0 | nothing to compose yet | Secondary counters (doom-loop halts avoided, M5 lean, avoided rework) are reported alongside when populated — they are deliberately excluded from the multiplication (spec §1). diff --git a/research/HISTORICAL_EDITIONS.md b/research/HISTORICAL_EDITIONS.md new file mode 100644 index 00000000..0ad05f00 --- /dev/null +++ b/research/HISTORICAL_EDITIONS.md @@ -0,0 +1,120 @@ +# Historical editions of the research papers + +> ⚠️ **Historical, pre-correction editions.** Every PDF listed here predates the 2026-09-21 and +> 2026-09-26 corrections. It is kept so that what was published can still be read and cited, not +> as the current text. The corrected sources are the HTML and LaTeX files named in the table; +> read those. + +## Status on 2026-09-26: not regenerated + +A re-render of the three HTML papers was attempted on 2026-09-26 and **not committed**: + +1. **Figures need resolving.** The HTML sources reference every figure through a + `{{artifact:…}}` placeholder, which no browser resolves, so a plain render has no figures. The + map below resolves them; it was checked against the images embedded in the old PDFs. +2. **The Qur'anic text could not be verified.** All three HTML papers carry Qur'anic Arabic. When + the white paper and the synthesis were rendered in the environment available that day, Chromium + set their verse text in three fallback fonts at once (DejaVu Sans for most glyphs, Liberation + Serif and FreeSerif for the rest); mixing fonts inside a word can break letter joining and mark + placement, and the rendered pages could not be inspected by eye. A PDF whose sacred text has not + been checked is not published as the new edition. +3. **The refutation paper needs TeX.** `empirical-refutation/paper.pdf` is built from + `paper/main.tex` with a TeX Live toolchain; none was available. + +## The editions + +Each edition stays retrievable byte for byte at the pinned commit +`d2abfa69fb77531199ffc67c5c076b524af69040`: +`git show d2abfa69fb77531199ffc67c5c076b524af69040: > edition.pdf`, or +`git cat-file -p ` with the blob below. + +| PDF (historical, pre-correction) | Git blob at `d2abfa6` | sha256 (prefix) | Bytes | Last changed in | Corrected source | +| --- | --- | --- | --- | --- | --- | +| `research/formal-synthesis/substrate_synthesis.pdf` | `2e17362fc62d9f32b1083f17a0ac865704244ef6` | `644e28d0f8d1cbd3…` | 835072 | `5e60069` (2026-08-14) | `research/formal-synthesis/substrate_synthesis.html` | +| `research/empirical-refutation/extended_preprint.pdf` | `74f74ae08612bb9dc11038f651f923e0730b8bfc` | `8d2ec1091d17f0ba…` | 791707 | `9ebe256` (2026-09-20) | `research/empirical-refutation/extended_preprint.html` | +| `research/empirical-refutation/paper.pdf` | `f94a727cec7f84ac197057f3f22cde0fb09f28b1` | `a5001d8fc59a1b4a…` | 830746 | `c5fb041` (2026-09-20) | `research/empirical-refutation/paper/main.tex` | +| `research/cognitive-substrate/cognitive_substrate_whitepaper.pdf` | `44ce7bbd4a7bc1e6220f162074c9b473c6287e7b` | `599e626ba24958c6…` | 1816584 | `e6e6de7` (2026-09-20) | `research/cognitive-substrate/cognitive_substrate_whitepaper.html` | +| `docs/cognitive-substrate/cognitive_substrate_whitepaper.pdf` (byte-identical copy) | same blob as the row above | same | 1816584 | — | the same HTML, copied to `docs/cognitive-substrate/` | + +The copies of `repro/paper/main.tex` and `repro/paper/paper.pdf` inside +`empirical-refutation/replication_package.tar.gz` (git blob `50bd453a30dad5d8ca3369129f8015fd4524fa81`) +are also left exactly as published. The paper PDF was built with pdfTeX (TeX Live 2026) and the ACM +`acmart` class, as its own metadata records. + +## Figure map + +Each `{{artifact:}}` placeholder in the HTML sources (13 in all: 3 in the synthesis, 3 in the +preprint, 7 in the white paper), the figure file under `research/` it stands for, and whether that +file's pixel size matches the image embedded in the old PDF: + +| Placeholder id | Figure (under `research/`) | Size (px) | Matches the old PDF | +| --- | --- | --- | --- | +| `art_5f049677-4c3c-40f2-8905-dd01c966e9ae` | `formal-synthesis/figures/schematic_duality.png` (synthesis and preprint, Figure 1) | 1366 × 1046 | yes | +| `art_f0decf80-d016-496f-8032-f7b2e73e71b2` | `formal-synthesis/figures/schematic_taskloop.png` (synthesis and preprint, Figure 2) | 1607 × 1092 | yes | +| `art_5d076ce3-0f54-4394-9a77-f70a336ca843` | `formal-synthesis/figures/schematic_convergence.png` (synthesis, Figure 8) | 2460 × 1539 | yes | +| `art_712fac51-fe17-4ed6-80f0-9dd42bf42758` | `empirical-refutation/figures/fig_repair_beforeafter.png` (preprint, Figure 3) | 3142 × 1383 | yes | +| `art_e2776474-3d1e-48c6-9490-55d4d927a301` | `cognitive-substrate/figures/schematic_loop.png` (white paper, Figure 1) | 3003 × 1439 | yes | +| `art_d2be1b53-86ce-4069-b2aa-5be59836598e` | `cognitive-substrate/figures/schematic_system.png` (white paper, Figure 2) | 2847 × 1840 | yes | +| `art_8d9fa6dd-3554-49c7-9e76-ba667544a622` | `cognitive-substrate/figures/schematic_extended.png` (white paper, Figure 3) | 1483 × 931 | yes | +| `art_07bb9186-5e55-44f6-af3f-dde83d6b9e65` | `cognitive-substrate/figures/impact_graph.png` (white paper, Figure 4) | 1900 × 1326 | yes | +| `art_392e293d-be93-4efe-81c1-e9612a711ac4` | `cognitive-substrate/figures/eval_precision_recall.png` (white paper, Figure 5) | 2300 × 918 | **no** — the old PDF embeds a 1921 × 842 raster, so the repository's file is a different render of this figure; compare the two before publishing | +| `art_5b206b7b-c90e-417f-b1e8-48b0ec389cb8` | `cognitive-substrate/figures/schematic_router_loop.png` (white paper, Figure 6) | 1537 × 838 | yes | +| `art_ac78be07-be03-4560-bb9f-f5fe2f16d7ef` | `cognitive-substrate/figures/router_eval.png` (white paper, Figure 7) | 1719 × 732 | yes | + +## Rendering a new edition + +1. Work in a scratch directory outside the repository and install `playwright-core` there, never + in the repository: `npm init -y && npm install playwright-core`. Point it at an installed + Chromium (`executablePath`). +2. Install a font with full Qur'anic coverage (a Naskh face such as Amiri or Scheherazade New) and + make it the first `font-family` for `.quran .ar` in the render, so one font sets each verse. +3. Replace every `{{artifact:}}` with its figure from the map (a `data:image/png;base64,…` + URI keeps the render self-contained), render A4 with a header and footer stamp + `edition · source sha256 · forgekit `, + and confirm every image loaded and no request failed. +4. **Inspect by eye** every page that carries a figure or a verse card. Only then replace the PDF, + re-copy the white paper to `docs/cognitive-substrate/` (`node scripts/claims-status.mjs + --sync-copies`), and add a row here recording the replaced edition's git blob and the new + edition's source hash. + +A render script that does steps 1 and 3 (it resolved all 13 figure placeholders across the three +papers with no failed request on 2026-09-26): + +```js +// node render.mjs (run from the scratch directory) +import { createHash } from "node:crypto"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { chromium } from "playwright-core"; + +const FIGURES = { /* "art_…": "research/…/figures/….png", one entry per row of the map above */ }; +const [repo, rel, out] = process.argv.slice(2); +const raw = readFileSync(path.join(repo, rel)); +const sha = createHash("sha256").update(raw).digest("hex").slice(0, 12); +const version = JSON.parse(readFileSync(path.join(repo, "package.json"), "utf8")).version; +const stamp = `edition ${new Date().toISOString().slice(0, 10)} · source sha256 ${sha} · forgekit ${version}`; +let html = raw.toString("utf8"); +for (const [id, fig] of Object.entries(FIGURES)) { + const uri = `data:image/png;base64,${readFileSync(path.join(repo, fig)).toString("base64")}`; + html = html.split(`{{artifact:${id}}}`).join(uri); +} +const browser = await chromium.launch({ executablePath: process.env.CHROMIUM }); +const page = await browser.newPage(); +const failed = []; +page.on("requestfailed", (r) => failed.push(r.url())); +await page.setContent(html, { waitUntil: "networkidle" }); +const unresolved = (html.match(/\{\{artifact:[^}]+\}\}/g) || []).length; +const broken = await page.evaluate(() => [...document.images].filter((i) => !i.naturalWidth).length); +if (unresolved || broken || failed.length) throw new Error(`figures: ${unresolved} unresolved, ${broken} broken, ${failed.length} failed`); +const line = (s) => `
${s}
`; +await page.pdf({ + path: out, format: "A4", printBackground: true, displayHeaderFooter: true, + margin: { top: "18mm", bottom: "18mm", left: "14mm", right: "14mm" }, + headerTemplate: line(stamp), + footerTemplate: line(`${stamp} · page / `), +}); +await browser.close(); +``` + +For `paper.pdf`, build `paper/main.tex` with TeX Live (`pdflatex` and `bibtex`, ACM `acmart` +class) and stamp the same fields in the PDF metadata or a footnote. diff --git a/research/README.md b/research/README.md index 6cac916d..8eea9665 100644 --- a/research/README.md +++ b/research/README.md @@ -1,56 +1,122 @@ # Research -The full research programme behind forgekit: a theory of what a frozen language model -structurally lacks, an architecture that supplies it, two runnable prototypes, and — most -importantly — a pre-registered empirical evaluation that **refuted the prototypes' headline -claims**. +The full research programme behind forgekit: an account of what a language model with frozen +weights does not guarantee on its own, an architecture that supplies it, two runnable +prototypes, and — most importantly — a pre-registered empirical evaluation that **refuted the +prototypes' headline claims**. Read in this order. The later work corrects the earlier work, and the corrections are the -most useful part. +most useful part. Every load-bearing headline below also has a row, with its status and the +evidence behind it, in the machine-readable claim registry +[`docs/status/claims.json`](../docs/status/claims.json), rendered as a table in +[`docs/status/README.md`](../docs/status/README.md). ## Start here: what is actually true | | Claimed (self-built demos) | Measured (real data) | |---|---|---| | Impact oracle recall | 1.00 | **0.022** — `grep` with no graph beats it ~10× on F1 | -| Router/gate F1 | 1.00 | **0.37** on 80 real GitHub issues/PRs | -| Cost saving | +62.1% | **−20.2%** — routing costs *more* than always-premium | +| Router/gate: gate F1 (should-ask) | 1.00 | **0.37** on 80 real GitHub issues/PRs | +| Router/gate: cost saving vs always-premium | +62.1% | **−20.2%** — routing costs *more* than always-premium | -Per output a judge accepted, the router cost $1.06 against always-premium's $1.76, but -only 6 and 3 of 64 outputs were accepted, so that comparison is not stable; 58 of the 64 -tasks failed at every tier, which is why escalation made routing cost more overall. +Success in the router rows means **judge-accepted**: a model judge accepted the output. No +held-out task admitted execution-based verification, so `tests_passed`, `human_accepted` and +`deployed_without_revert` were never measured. Per judge-accepted output the router cost $1.06 +against always-premium's $1.76, but only 6 and 3 of the 64 non-halted tasks were judge-accepted, +so that ratio is not stable; 58 of the 64 tasks failed at every tier, which is why escalation made +routing cost more overall ($6.3582 against $5.2893, 20.21% more). The judge was also the mid-tier +executor, and the "second labelling pass" is the same model with a reworded prompt (n = 30: halt +κ 0.5161, tier κ 0.8919), so κ measures self-consistency, not agreement with a human. +(Corrected 2026-09-26: the first sentence read "Per output a judge accepted, …" and named neither +what the judge's acceptance is not nor who the judge was; the row was labelled "Router/gate F1".) -After diagnosing and repairing two defects, with numeric parameters frozen before the -held-out repositories were touched: recall **0.653**, F1 **0.416**, a point estimate above -`grep`'s 0.371 for the first time. That the repaired oracle *beats* grep is **not -established**: the three held-out repositories all favour it, but three out of three is a -one-sided sign-test p of 0.125, pytest supplies 71% of the held-out pairs, the file-level -intervals overlap, and the choice of which relations to add was made on all nine -repositories. (Corrected 2026-09-21; earlier versions called it "a real but narrow win".) +After diagnosing and repairing two defects, with numeric parameters frozen before the held-out +repositories were touched: at the pre-registered canonical threshold 0.02, recall **0.653** and +F1 **0.416**, a point estimate above `grep`'s 0.371 for the first time (paired ΔF1 about +0.044). +That the repaired oracle *beats* grep is **not established**. Per-repository counts exist only at +threshold 0.10, where the pooled ΔF1 is +0.0565 and all three held-out repositories favour the +oracle, but three out of three is a one-sided sign-test p of 0.125; pytest supplies 71.3% of the +held-out pairs; the file-level intervals overlap; and the choice of which relations to add was made +after diagnosing all nine repositories — an architecture-selection channel into the nominal test +set, which limits the unseen-repository claim without erasing the measured gain. Numbers at 0.02 +and 0.10 are never mixed in one comparison. (Corrected 2026-09-21; earlier versions called it "a +real but narrow win". Thresholds separated 2026-09-26.) The general lesson, demonstrated on our own work: **a self-built demonstration can overstate field performance by more than an order of magnitude, and careful caveating does not convert a demonstration into evidence.** +## What the impact study measured — and what it did not + +*(Added 2026-09-26, after a second external review recomputed the archived results.)* + +The archived counts reproduce: 801 labelled files, 759 evaluated after the pre-registered cap of +200 files per repository, nine repositories, 20,144 mirrored labelled pairs. The original oracle's +pooled precision / recall / F1 is **0.3982 / 0.0220 / 0.0416**; grep's is **0.3535 / 0.5732 / +0.4373**. A repository-cluster bootstrap (20,000 draws, seed 1234) gives oracle F1 **[0.0010, +0.0927]**, grep **[0.3807, 0.5394]**, and grep minus oracle **[0.3422, 0.5174]** — figures +recomputed by the 2026-09-26 external review and reproduced by +[`recompute_corrections.py`](recompute_corrections.py) §5, which also prints the repository-level +view: macro F1 0.0220 for the oracle against 0.4947 for grep, with grep ahead in 9 of 9 +repositories. The negative result is well supported within this archived corpus. + +What it is a result *about* needs stating as carefully as the numbers: + +- **Co-change is a proxy.** Two files that changed in the same commit are *historically related + edits*, not proof of semantic necessity or of test breakage; conversely a dependency graph is not + a full co-change graph. The study measured one task: (a) predicting co-edited files. The other + task an impact tool is used for — (b) selecting the tests that detect a behaviour regression — + was not measured, and results for the two should always be reported separately. +- **Pairs are not independent.** Every ground-truth pair is mirrored (counted from both ends) and + files share repositories, so file- or pair-level resampling overstates precision. Uncertainty is + reported at the repository level, and macro results sit beside pooled ones. +- **The Node graph is not the evaluated oracle.** The shipped `forge impact` / `src/atlas.js` is a + regex-derived, multi-language code graph that ports the two repairs; it is not the Python AST + oracle the study evaluated, and the study's numbers are not its numbers. Its own measurement is a + six-case, self-labelled fixture in [`reports/benchmarks.md`](../reports/benchmarks.md). + +### Next study (pre-declared shape) + +The nine-repository archive is a reproducibility starter, not a fresh holdout. The next impact +study freezes the parser and relation design **before** acquiring a new repository set or time +split; includes runtime coupling, configuration changes, dynamic imports and languages other than +Python; predeclares relation budgets so that widening predictions cannot win merely by returning +most files; and reports review-cost metrics — files reviewed per true affected file, and the +missed-regression rate — beside F1, for co-edited-file prediction and regression-test selection +separately. + ## The four layers ### 1. [`cognitive-substrate/`](cognitive-substrate/) — the theory -The originating argument: an LLM is a frozen map `y = f_θ(x)` with three properties — -statelessness, frozen parameters, bounded context — which structurally deny it five faculties -(memory, learning, imagination, self-correction, impact-awareness). The remedy is an external -stateful architecture, not better prompting. +The originating argument: an LLM is a map `y = f_θ(x)` with frozen parameters, no state between +calls and a bounded context, and a coding agent built on it lacks five faculties (memory, +learning, imagination, self-correction, impact-awareness) unless something outside supplies them. +Stated precisely, what is missing is a set of guarantees — no durable state across independent +invocations, a bounded context, no automatic parameter update, and unreliable self-verification +without external evidence. Prompting does change behaviour inside a context (in-context adaptation; +Brown et al., 2020, [arXiv:2005.14165](https://arxiv.org/abs/2005.14165)); the substrate is a tested +way of supplying persistence and verification, not the only logically possible architecture. +(Corrected 2026-09-26: this paragraph said the three properties "structurally deny it five +faculties" and that "the remedy is an external stateful architecture, not better prompting".) -- `cognitive_substrate_whitepaper.pdf` — the *Theory → Evidence → Build-Map* edition (48pp); - the `.html` edition carries the 2026-09-21 corrections, the PDF predates them +- `cognitive_substrate_whitepaper.pdf` — the *Theory → Evidence → Build-Map* edition (48pp). + **Historical, pre-correction edition** (git blob `44ce7bb`); the `.html` edition is the corrected + source and carries the 2026-09-21 and 2026-09-26 corrections. - `EXECUTIVE_SUMMARY.md` — one-page entry point, **carries a status banner: its prototype numbers are refuted** - `literature/` — the gap map and 32 graded references behind each faculty claim - `evidence/` — twelve load-bearing industry statistics independently re-grounded and graded `confirmed` / `vendor-reported` / `unverifiable`, plus an ecosystem map of what the 2026 Claude-Code stack already solves. Three widely-repeated statistics were caught as - misattributed and dropped. + misattributed and dropped. Since 2026-09-26 the evidence map also grades claim support, study + design, independent replication and transfer scope separately; the original grades mainly + confirm that a source exists and says what is quoted. - `quranic-lens/` — the fourteen-mapping ethical-epistemic reading used as a *design lens*: it names which safeguards are obligatory rather than optional. It is framing, never - technical authority; no verse is offered as proof of an engineering claim. + technical authority; no verse is offered as proof of an engineering claim. The Arabic source + text, the translation, tafsir and the author's design analogy are labelled separately, and the + lens's operational content is the discipline *do not assert without evidence* — the claim + registry, verifier events and visible uncertainty — not any algorithm's correctness, catch rate + or uniqueness. - `sources/` — the primary documents the evidence layer was graded against - `figures/` — the architecture schematics and prototype evaluations @@ -62,7 +128,11 @@ two-layer duality: the silent-miss residual is `(1 − p) × P(no deterministic check fires | miss)`, so where each factor is bounded away from zero, neither layer alone reaches a small residual. Since the 2026-09-21 corrections this is stated as a bound over an explicit `(p, q)` region, not as a proof that -neither layer suffices, and the checks multiply only if they fire independently. +neither layer suffices, and the checks multiply only if they fire independently. Since the +2026-09-26 corrections the reachable residual is a minimum over the *jointly* feasible `(p, q)` +pairs — separately maximal `p` and `q` need not be attainable under one policy, so +`(1 − p_max)(1 − q_max)` is only a lower bound — and a caught miss is no longer read as a +completed task. **Priority note:** prior-art review found this composition law is standard protection-layer algebra, and two concurrent preprints derive a strictly more general Bayesian form weeks @@ -70,6 +140,9 @@ earlier. Priority is conceded in the refutation paper's related work and, since 2026-09-21 corrections, in the synthesis and the extended preprint as well (before that they still said "this paper proves"). What survives is that both preprints are simulation-only. +- `substrate_synthesis.pdf` — **historical, pre-correction edition** (git blob `2e17362`); the + corrected source is `substrate_synthesis.html`. + ### 3. [`empirical-refutation/`](empirical-refutation/) — the measurement The pre-registered evaluation that overturned the claims above, the diagnosis of *why*, and the repair. Includes a replication package with the frozen pre-registration, mined ground @@ -81,10 +154,44 @@ Also corrects a theoretical claim: perfect recall was inferred from a completene but such a theorem guarantees completeness only *relative to the relation* the closure runs over — it says nothing about whether that relation contains the edges that matter. +- `replication_package.tar.gz` — the archive **exactly as published** (git blob `50bd453`); its + copies of `paper/main.tex` and `paper.pdf` predate the corrections. Corrected summary: the + README's Corrections sections; every corrected number is recomputed from it by + `recompute_corrections.py`. +- `paper.pdf` and `extended_preprint.pdf` — **historical, pre-correction editions** (git blobs + `f94a727`, `74f74ae`); the corrected sources are `paper/main.tex` and `extended_preprint.html`. + ### 4. [`python-prototypes/`](python-prototypes/) — the code -`impact_oracle/` and `router_gate/`, runnable with their own test suites. The **repaired** -oracle ships inside the refutation's replication package rather than replacing the version -here, so swapping it in stays a deliberate decision. +`impact_oracle/` and `router_gate/`, runnable with their own test suites. The in-tree +`impact_oracle/` **is the repaired (v2) oracle**: both repairs are in its source, with their +frozen parameters as module defaults, and its suite is 49 tests (36 demo-package tests plus 13 +regression tests for the two repairs). `ImpactOracle(wm, sibling_enabled=False, +forward_enabled=False)` reproduces the refuted reverse-only traversal, and the untouched as-shipped +v1 package is archived in the replication tarball. (Corrected 2026-09-26: this paragraph said the +repaired oracle "ships inside the refutation's replication package rather than replacing the +version here, so swapping it in stays a deliberate decision"; the swap had already been made.) + +## Prior art, and what is (and is not) claimed + +*(Added 2026-09-26.)* External memory, feedback-driven improvement and structured agent control +all have clear prior art. **CoALA** (Sumers et al., 2023, +[arXiv:2309.02427](https://arxiv.org/abs/2309.02427)) organises language agents into modular +memory, action and decision procedures; **Reflexion** (Shinn et al., 2023, +[arXiv:2303.11366](https://arxiv.org/abs/2303.11366)) improves agents through linguistic feedback +and an episodic memory buffer, with no weight updates; GPT-3's few-shot evaluation +([arXiv:2005.14165](https://arxiv.org/abs/2005.14165)) already measured adaptation through text +alone. That prior art does not make forgekit unoriginal as a product, but it limits what the broad +architecture can claim. The defensible framing is: + +> **a portable implementation of evidence-weighted coding-agent memory and checks, with +> empirical evaluation of trust failure modes.** + +Novelty is claimed only for a specific protocol, invariant, evaluation result or integration that +survives an explicit comparison with that prior art. The "five faculties" are a useful +decomposition, not a proof that these five are necessary or that an external stateful architecture +is the only way to supply them; and the "convergence" of the theory, forgekit and its sibling +projects is consistency within one author's work, not independent confirmation — the same care the +programme already applied when it conceded priority for the protection-layer equation. ## How this programme tries to stay honest @@ -98,6 +205,15 @@ Where this falls short is stated too: the pre-registration and parameter freezes self-administered with no external timestamping authority, so a reader can verify internal consistency and the amendment trail but must take the ordering on trust. +### Four kinds of reproducibility, kept apart + +| Kind | Status | +|---|---| +| **Source availability** | The papers' corrected sources (HTML, LaTeX), both Python prototypes, the replication archive and the recomputation script are all in this directory. | +| **Calculation reproducibility** | Available. [`recompute_corrections.py`](recompute_corrections.py) (standard library only) recomputes every corrected statistic from the archived results, and asserts the Theorem D sanity checks with no data at all (`--theorem-checks`); CI runs both, and both prototypes' test suites, since `aedddf5`. The universal router's shipped prior also refits exactly from pinned public data with `bench/universal-router/reproduce.sh` (2026-09-26, the project's own run). | +| **Pipeline reproducibility** | **Not available.** The historical mining pipeline (cloning, commit filtering, labelling, model calls) is not shipped as an entry point here, so new histories cannot be mined with one command; and the universal router's run-4 held-out benchmark ran in an external harness (harness-bench) that is not shipped either — see [`docs/UNIVERSAL_ROUTING.md`](../docs/UNIVERSAL_ROUTING.md). | +| **Independent external replication** | None of the research results has been replicated by an independent team. The 2026-09-21 and 2026-09-26 reviews recomputed archived numbers; they did not re-mine repositories or re-run model calls. | + ## Corrections (2026-09-21) An external deep review of this repository (2026-09-21) recomputed the research statistics @@ -126,10 +242,40 @@ mkdir rp && tar -xzf research/empirical-refutation/replication_package.tar.gz -C python research/recompute_corrections.py rp/repro ``` -**Stale PDFs.** These PDFs predate the corrections and could not be rebuilt here (the HTML -editions were rendered with WeasyPrint, the paper with a TeX Live toolchain; neither was -available): `formal-synthesis/substrate_synthesis.pdf`, +## Corrections (2026-09-26) + +A second external deep review (2026-09-26, pinned at commit +`d2abfa69fb77531199ffc67c5c076b524af69040`) recomputed the archived results again and read the +papers' framing against the code. The counts reproduced exactly again. Changes, each marked in +place in its paper with `[corrected 2026-09-26]` and listed there with the original wording: + +- **Formal synthesis and extended preprint** — the range statement of Theorem D no longer combines + separately maximal `p` and `q` (counterexample: policies `(0.5, 0.9)` and `(0.9, 0.1)` leave 0.05 + and 0.09, while the separate maxima suggest 0.01); the equality condition for + `1 − (1 − ε)ⁿ` is every `rᵢ = ε`, not independence alone; a new §5.4 separates silent-miss + probability from completed-task rate and lists what to measure; the frozen-map premise is stated + as the guarantees it removes; prior art (CoALA, Reflexion) is named and the byline no longer + calls the three bodies of work "independently-developed". The counterexample, the equality + condition and the 400× correction are asserted by `python3 research/recompute_corrections.py + --theorem-checks`. +- **Whitepaper** — the "cannot learn / imagine / self-correct" framing marked as broader than the + missing guarantees; prior art added to §11; the Qur'anic lens's text, translation, tafsir and + design analogy labelled separately, with its operational scope stated; METR's 19% slowdown + scoped to its 16 developers, 246 tasks and early-2025 tools, with METR's + [February 2026 update](https://metr.org/blog/2026-02-24-uplift-update/). +- **This README and the prototype READMEs** — the repaired oracle's location, the 0.02 / 0.10 + thresholds, judge-accepted versus executed success, the unit of the impact study, and the + reproducibility table above. + +**Stale PDFs — historical, pre-correction editions.** `formal-synthesis/substrate_synthesis.pdf`, `empirical-refutation/extended_preprint.pdf`, `empirical-refutation/paper.pdf`, -`cognitive-substrate/cognitive_substrate_whitepaper.pdf`, and the copy in -`docs/cognitive-substrate/`. The copies of `paper/main.tex` and `paper.pdf` inside -`replication_package.tar.gz` are left as published. Read the HTML and LaTeX sources. +`cognitive-substrate/cognitive_substrate_whitepaper.pdf` and its byte-identical copy in +`docs/cognitive-substrate/` predate both sets of corrections. Read the HTML and LaTeX sources, +which carry them. A re-render was attempted on 2026-09-26 and **not** committed: the HTML sources +reference their figures through `{{artifact:…}}` placeholders that a browser cannot resolve, all +three HTML papers carry Qur'anic Arabic whose typesetting could not be checked by eye in that +environment, and a PDF whose figures or sacred text cannot be verified is not published as the new +edition. The paper PDF needs a TeX toolchain that was not available. Each edition's git blob, the +pinned commit where it stays retrievable, and a faithful render recipe (including the +figure-placeholder map) are in [`HISTORICAL_EDITIONS.md`](HISTORICAL_EDITIONS.md). The copies of +`paper/main.tex` and `paper.pdf` inside `replication_package.tar.gz` are left as published. diff --git a/research/cognitive-substrate/EXECUTIVE_SUMMARY.md b/research/cognitive-substrate/EXECUTIVE_SUMMARY.md index 0938a0af..f6d12774 100644 --- a/research/cognitive-substrate/EXECUTIVE_SUMMARY.md +++ b/research/cognitive-substrate/EXECUTIVE_SUMMARY.md @@ -7,7 +7,7 @@ > | Claim below | Measured on real data | > |---|---| > | Impact oracle recall **1.00** | **0.022** (9 OSS repos; 801 labelled files, 759 evaluated); `grep` beats it ~10× on F1 | -> | Router/gate F1 **1.00**, cost saving **+62.1%** | F1 **0.37**; cost saving **−20.2%** (routing costs *more* than always-premium; per judged-correct output $1.06 vs $1.76, from only 6 and 3 correct outputs of 64) | +> | Router/gate F1 **1.00**, cost saving **+62.1%** | F1 **0.37**; cost saving **−20.2%** (routing costs *more* than always-premium; per judge-accepted output $1.06 vs $1.76, from only 6 and 3 judge-accepted outputs of 64; the judge is a model, not executed tests) | > > The theory sections remain the programme's working framework. The *numbers* here do not. A repair > raised recall to 0.653 and F1 to 0.416, a point estimate above grep's 0.371, documented in the @@ -25,6 +25,17 @@ **One-line thesis:** The faculties a coding agent lacks — memory, learning, imagination, self-correction, impact-awareness — are not gaps in the model's *knowledge* but structural consequences of what a frozen transformer *is* (a stateless map `y = f_θ(x)`, fixed weights, bounded window). They cannot be prompted or tooled away; they can only be supplied by **re-wrapping the input→process→output loop** into a closed, stateful cycle around the frozen model. +> *Corrected 2026-09-26:* the thesis above is broader than its argument. Frozen weights rule out +> weight updates during use, not all adaptation: examples, retrieved facts and feedback in the +> context change behaviour with no gradient step (Brown et al., 2020, +> [arXiv:2005.14165](https://arxiv.org/abs/2005.14165)). What a bare model lacks is a set of +> guarantees — no durable state across independent invocations, a bounded context, no automatic +> parameter update, and unreliable self-verification without external evidence — and the substrate +> is one tested way of supplying persistence and verification, not the only possible architecture. +> Prior art for the broad architecture includes CoALA ([arXiv:2309.02427](https://arxiv.org/abs/2309.02427)) +> and Reflexion ([arXiv:2303.11366](https://arxiv.org/abs/2303.11366)); see the white paper's +> Corrections (2026-09-26). + **What v2 adds.** The first edition argued the five faculties from first principles and prototyped the one that is buildable today. This edition (1) **grounds the argument in the field's own evidence** — twelve load-bearing pain-point statistics independently re-grounded from primary sources and graded *confirmed / vendor-reported / unverifiable*; (2) adds **six metacognitive mechanisms** the frozen loop also lacks (routing, assumption gate, decomposition, goal-anchoring, anti-over-engineering, inline verification); (3) **maps all eleven capabilities against the real 2026 Claude-Code stack**, marking each solved / partial / residual-gap so we say clearly *what not to build*; and (4) ships a **second runnable prototype** — a complexity-aware router + assumption gate, evaluated live on real models. > **Governing discipline (the user's, adopted throughout):** *AI output is mathematically-calculated probability — non-deterministic, and never blindly trusted.* Every claim in this package is graded by how well it is sourced; every prototype decision is a transparent, attributable rule rather than another opaque model call; and trust is always earned by an **external** check, never asserted by the model. diff --git a/research/cognitive-substrate/cognitive_substrate_whitepaper.html b/research/cognitive-substrate/cognitive_substrate_whitepaper.html index 95460e9b..615495dc 100644 --- a/research/cognitive-substrate/cognitive_substrate_whitepaper.html +++ b/research/cognitive-substrate/cognitive_substrate_whitepaper.html @@ -65,6 +65,7 @@ .quran .map{font-size:.92rem; color:var(--ink); margin-top:.6em; padding-top:.6em; border-top:1px dotted #d8ccae;} .quran .map b{color:var(--quran);} .quran .grounding{font-size:.74rem; color:var(--faint); margin-top:.5em; font-family:monospace;} + .quran .lbl{font-family:sans-serif; font-size:.64rem; text-transform:uppercase; letter-spacing:.08em; color:var(--faint); margin:.55em 0 .1em;} .lit{background:var(--litbg); border:1px solid #d9e2ea; border-radius:4px; padding:6px 14px; margin:1.1em 0; font-size:.9rem;} .callout{border:1px solid var(--rule); border-radius:5px; padding:14px 20px; margin:1.4em 0; background:#fcfcfc;} .callout.key{border-left:3px solid var(--oracle); background:#f2f9f8;} @@ -96,13 +97,13 @@

A Cognitive Substrate for Coding Agents

-
Status: both prototype results in this edition were refuted · corrections 2026-09-21
-

This edition was written before any real-repository evaluation existed. A later pre-registered evaluation (research/empirical-refutation/) overturned both prototype claims: the impact oracle’s recall was 0.022, not 1.00, on 759 files in nine open-source repositories, and on 80 held-out tasks the router’s total spend was 20.2% higher than always using the premium tier, not 62.1% lower. An external review (2026-09-21) also found a misquoted statistic, a wrong worst-case cost, and an inconsistency between Eq. (1) and M2. Corrections are made in place, marked [corrected 2026-09-21] or [refuted], and listed with the original wording in Corrections. The theory sections remain the programme’s working framework; the prototype numbers do not. The PDF edition predates these corrections.

+
Status: both prototype results in this edition were refuted · corrections 2026-09-21 and 2026-09-26
+

This edition was written before any real-repository evaluation existed. A later pre-registered evaluation (research/empirical-refutation/) overturned both prototype claims: the impact oracle’s recall was 0.022, not 1.00, on 759 files in nine open-source repositories, and on 80 held-out tasks the router’s total spend was 20.2% higher than always using the premium tier, not 62.1% lower. An external review (2026-09-21) also found a misquoted statistic, a wrong worst-case cost, and an inconsistency between Eq. (1) and M2. Corrections are made in place, marked [corrected 2026-09-21] or [refuted], and listed with the original wording in Corrections. The theory sections remain the programme’s working framework; the prototype numbers do not. A second review (2026-09-26) found that the “cannot learn / imagine / self‑correct” framing is broader than the missing guarantees it rests on, that the prior art for the architecture needed stating, that the Qur’anic lens mixed source text, translation and the author’s analogy without labels, and that the METR statistic needed its scope; those are listed in Corrections (2026-09-26) and marked [corrected 2026-09-26]. The PDF edition predates both sets of corrections.

Abstract

-

A large language model at inference time is, mathematically, a fixed function y = fθ(x) with frozen parameters θ and a bounded input window. From this single fact, five apparent “cognitive” deficits of a coding agent follow as structural consequences, not incidental weaknesses: it cannot remember across sessions, cannot learn from outcomes, cannot imagine the consequences of an action before taking it, cannot reliably correct itself, and does not know what already exists in a codebase or what an edit will affect. We show that neither better prompting nor additional tools (skills, MCP servers) remove these deficits, because they leave fθ and the open‑loop pipeline intact. We then specify a cognitive substrate: an external architecture that keeps the LLM frozen but re‑wraps its input→process→output loop into a closed, stateful cycle over persistent stores — an episodic/semantic memory, an online‑updatable learning layer, a consequence simulator, a metacognitive verification gate, and a persistent structural model of the codebase — all under an explicit stewardship boundary. For each faculty we identify precisely what the existing literature solves and what residual gap remains for a coding agent. To turn the weakest‑evidenced claim into something testable, we build and evaluate the impact‑awareness faculty as a runnable prototype: a Codebase World‑Model that parses a repository into a persistent dependency graph, and an Impact Oracle that predicts the blast radius of a proposed edit. Against mutation‑derived ground truth on a ten‑file package we wrote, the oracle was the only method that missed no affected file (recall = 1.00 across five tested edits), where a text‑search baseline missed transitive dependents and an edited‑file‑only baseline missed 47% of impact. On nine real repositories its recall was 0.022, and text search beat it by an order of magnitude on F1. [refuted — see Corrections] Throughout, a Qur'anic epistemic lens supplies the design's vocabulary of obligation — know what exists before acting (2:31–32), verify before you act (49:6), pursue not that of which you have no knowledge (17:36), and hold what you can damage as a trust (33:72).

+

A large language model at inference time is, mathematically, a fixed function y = fθ(x) with frozen parameters θ and a bounded input window. From this single fact, five apparent “cognitive” deficits of a coding agent follow as structural consequences, not incidental weaknesses: it cannot remember across sessions, cannot learn from outcomes, cannot imagine the consequences of an action before taking it, cannot reliably correct itself, and does not know what already exists in a codebase or what an edit will affect. We show that neither better prompting nor additional tools (skills, MCP servers) remove these deficits, because they leave fθ and the open‑loop pipeline intact. [corrected 2026-09-26 — see Corrections] We then specify a cognitive substrate: an external architecture that keeps the LLM frozen but re‑wraps its input→process→output loop into a closed, stateful cycle over persistent stores — an episodic/semantic memory, an online‑updatable learning layer, a consequence simulator, a metacognitive verification gate, and a persistent structural model of the codebase — all under an explicit stewardship boundary. For each faculty we identify precisely what the existing literature solves and what residual gap remains for a coding agent. To turn the weakest‑evidenced claim into something testable, we build and evaluate the impact‑awareness faculty as a runnable prototype: a Codebase World‑Model that parses a repository into a persistent dependency graph, and an Impact Oracle that predicts the blast radius of a proposed edit. Against mutation‑derived ground truth on a ten‑file package we wrote, the oracle was the only method that missed no affected file (recall = 1.00 across five tested edits), where a text‑search baseline missed transitive dependents and an edited‑file‑only baseline missed 47% of impact. On nine real repositories its recall was 0.022, and text search beat it by an order of magnitude on F1. [refuted — see Corrections] Throughout, a Qur'anic epistemic lens supplies the design's vocabulary of obligation — know what exists before acting (2:31–32), verify before you act (49:6), pursue not that of which you have no knowledge (17:36), and hold what you can damage as a trust (33:72).

@@ -155,7 +157,7 @@

2 The root cause, formally

FacultyWhy it is structurally absentFollows from Memory (across sessions)By P1, nothing survives a turn but the token string; by P3, the string is bounded and lost at session end. There is no addressable store that outlives x.P1, P3 -Learning (from outcomes)By P2, no inference‑time event writes to θ. In‑context “learning” is real optimization14,15 but lives only inside the current x and vanishes with it (P1) — a simulation of learning, not learning.P2, P1 +Learning (from outcomes)By P2, no inference‑time event writes to θ. In‑context “learning” is real optimization14,15 but lives only inside the current x and vanishes with it (P1) — a simulation of learning, not learning. [corrected 2026-09-26]P2, P1 Imagination (simulate before acting)Equation (1) maps tokens to tokens. There is no separate forward model of “what happens to the world (or codebase) if I take action a” distinct from emitting more tokens; the model cannot roll out and score a hypothetical it does not also have to narrate.P1 Self‑correctionAny “check” the model runs is another evaluation of the same fθ with the same blind spots. There is no independent verifier inside Equation (1); the literature confirms intrinsic self‑correction is unreliable without an external signal.21P2 Impact‑awarenessBy P3, the model sees only the tokens in x. A million‑line repository does not fit; therefore it cannot know, unaided, what elsewhere depends on the symbol it is about to change.P3 @@ -165,6 +167,7 @@

2 The root cause, formally

Why prompting and tools do not close the gap

A better prompt changes x. More tools (skills, MCP servers, function calls) let the agent fetch new x or emit richer y. Both operate inside Equation (1) and leave P1–P3 untouched: the composed system is still a stateless map with frozen weights and a bounded window. A tool call retrieves a document into context, but nothing decides what was worth keeping, consolidates it, or updates the agent's priors for next time. The deficits are properties of the loop shape — open, memoryless, one‑directional — not of the model's knowledge. To remove them you must change the shape of the loop, which is precisely what an external substrate can do while θ stays frozen.

+

[corrected 2026-09-26] This callout overstates its case. Prompting and tools do change behaviour: examples, retrieved facts, feedback and extra computation placed in the context adapt a frozen model with no gradient update (Brown et al., 2020). What they do not supply by themselves is four guarantees: durable state across independent invocations, context beyond the window, automatic parameter update from outcomes, and reliable self‑verification without external evidence. The substrate is one tested way of supplying persistence and verification around the model, not the only logically possible architecture.

This reframing is the paper's pivot. If the deficits came from the loop shape, then the remedy is to re‑wrap the loop: keep fθ exactly as it is, and surround it with state and update so that the composite system is no longer memoryless, no longer open, and no longer blind beyond W. Figure 1 states the whole thesis in one picture.

@@ -246,6 +249,8 @@

4.1 What the strongest evidence confirms

has no calibrated sense of its own uncertainty (mechanism M2 below), and it is why “the model felt confident” is not evidence of anything.

+

[corrected 2026-09-26] Scope: this is one randomized trial of 16 experienced developers on 246 tasks in repositories they knew, with early‑2025 tools. It is evidence about that setting, not a universal 2026 productivity coefficient in either direction. METR’s February 2026 update (metr.org/blog/2026-02-24-uplift-update) explains why selection effects complicate newer estimates.

+

The trend evidence is equally well‑sourced. Stack Overflow's 2025 survey of more than 49,000 developers records trust in AI accuracy falling from 40 % to 29 % even as adoption rose to 84 %, with the top‑ranked frustration — cited by 66 % — being code @@ -312,38 +317,51 @@

5 The Qur'anic epistemic lens

How this lens is used — and how it is not

The Qur'an is used here as a framing lens and ethics source, never as technical authority for an engineering claim. No verse is cited to prove that an algorithm works or that a data structure is correct — those claims stand on their engineering merits alone (§3, §8). What the lens supplies is threefold: (1) a precise vocabulary of obligation for what an agent that acts on real systems owes — to truthfulness, to verification, to stewardship; (2) a hierarchy of knowledge (‘ilm → fahm → ḥikma: knowledge → understanding → wisdom) that motivates a layered memory architecture rather than a flat vector store; and (3) ethical constraints on autonomy that translate into concrete safeguards. Where a mapping is marked load‑bearing, the concept motivates a specific design decision (e.g. a mandatory, not optional, verification gate); where marked metaphor, it is illustrative. Canonical text below is presented directly and attributed; it is not paraphrased. Arabic and translations were retrieved from quran.ai; the full 14‑row mapping table is in the appendix.

+

[corrected 2026-09-26] How to read each card. Four layers are kept apart and labelled: the Arabic source text (clean Uthmani script; an ellipsis marks an abridgement, and the full verse is in quranic-lens/quran_lens.md); the English translation (M.A.S. Abdel Haleem); tafsir (classical commentary, Ibn Kathir), which these cards do not quote — the grounding line only records that it was consulted, and the companion file quotes it under its own label; and the author’s design analogy, which is the author’s engineering reading and neither a translation nor a commentary.

The lens earns its place because the deepest failure modes of an autonomous coding agent are not computational but epistemic and ethical: acting without knowing, trusting a report without checking it, and treating a granted capability as license. The Qur'anic vocabulary names these with unusual precision, and three verses in particular map so directly onto architectural decisions that they shaped the design rather than decorating it.

17:36 — lā taqfu · the root of impact‑awareness  [load‑bearing]
+
Arabic source text
وَلَا تَقْفُ مَا لَيْسَ لَكَ بِهِ عِلْمٌ ۚ إِنَّ السَّمْعَ وَالْبَصَرَ وَالْفُؤَادَ كُلُّ أُولَٰئِكَ كَانَ عَنْهُ مَسْئُولًا
+
English translation — M.A.S. Abdel Haleem
“Do not follow blindly what you do not know to be true: ears, eyes, and heart, you will be questioned about all these.”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. Before any mutation (write, delete, refactor), a pre‑action gate must establish what the agent actually knows: which entities the change affects, whether the relevant files/tests/dependents were actually read, and whether the predicted outcome rests on evidence rather than a pattern‑matched guess. Actions taken without verified knowledge are blocked, not merely flagged. The verse's own structure — “ears, eyes, heart… questioned about all these” — maps to an audit trail: every channel the agent used (what it read, inferred, assumed) is logged so the decision can be reconstructed and questioned. This is precisely the impact oracle of §8.
Grounded with quran.ai: fetch_translation(17:36, en-abdel-haleem); fetch_tafsir(17:36, en-ibn-kathir)
49:6 — tabayyun · the verification gate  [load‑bearing]
+
Arabic source text
يَا أَيُّهَا الَّذِينَ آمَنُوا إِن جَاءَكُمْ فَاسِقٌ بِنَبَإٍ فَتَبَيَّنُوا أَن تُصِيبُوا قَوْمًا بِجَهَالَةٍ فَتُصْبِحُوا عَلَىٰ مَا فَعَلْتُمْ نَادِمِينَ
+
English translation — M.A.S. Abdel Haleem
“Believers, if a troublemaker brings you news, check it first, in case you wrong others unwittingly and later regret what you have done.”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. The architecture places a verification step between receiving information (from context, tool output, or its own prior reasoning) and acting on it. The operative term tabayyun demands active investigation, not passive acceptance: before applying a fix based on an error report or its own diagnosis, the agent re‑reads the current file state, confirms the issue still exists, and checks the fix introduces no new breakage detectable by types or tests. The gate is architectural and mandatory — the verse's command is categorical, not conditional on the reporter's trustworthiness — which is exactly the design lesson the self‑correction literature reached empirically21: a real external check, not more self‑prompting.
Grounded with quran.ai: fetch_translation(49:6, en-abdel-haleem); fetch_tafsir(49:6, en-ibn-kathir)
2:31–32 — ta‘līm al‑asmā' · the world‑model  [load‑bearing]
+
Arabic source text
وَعَلَّمَ آدَمَ الْأَسْمَاءَ كُلَّهَا ... قَالُوا سُبْحَانَكَ لَا عِلْمَ لَنَا إِلَّا مَا عَلَّمْتَنَا
+
English translation — M.A.S. Abdel Haleem
“He taught Adam all the names [of things]… They said, ‘May You be glorified! We have knowledge only of what You have taught us.’”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. Knowledge begins with naming — identifying entities and their relations. The agent must hold a structured map of what exists in the codebase (files, functions, classes, dependencies), not a flat listing: knowing that A calls B, that module X depends on Y, that test T covers class C. The angels' admission — “we have knowledge only of what You have taught us” — is a startlingly exact description of the LLM's own situation under P3: it knows only what is in its window. The external world‑model supplies the “names” of entities that exceed context capacity.
Grounded with quran.ai: fetch_translation(2:31-32, en-abdel-haleem)
33:72 — al‑amāna · stewardship & bounded autonomy  [load‑bearing]
+
Arabic source text
إِنَّا عَرَضْنَا الْأَمَانَةَ عَلَى السَّمَاوَاتِ وَالْأَرْضِ وَالْجِبَالِ فَأَبَيْنَ أَن يَحْمِلْنَهَا ... وَحَمَلَهَا الْإِنسَانُ ۖ إِنَّهُ كَانَ ظَلُومًا جَهُولًا
+
English translation — M.A.S. Abdel Haleem
“We offered the Trust to the heavens, the earth, and the mountains, yet they refused to undertake it and were afraid of it; mankind undertook it — they have always been inept and foolish.”
+
Author’s design analogy — not translation, not tafsir
→ Design principle. An agent that can modify a codebase bears an amāna — accepted responsibility for something it can damage. The verse's structure is the design: the heavens declined the trust, recognizing its weight; the human bore it and is called ẓalūman jahūlā (given to wrong and ignorance). So the architecture (a) operates under least privilege — a granted capability is never blanket permission; and (b) assumes the agent will err and builds in reversibility (sandboxing, staged commits, rollback), audit, and scope bounds as structural safeguards. The trust is not “the agent is trustworthy”; it is “the agent has accepted accountability for a domain it can harm, and the architecture must respect that weight.” This is the governance boundary enclosing the entire system in Figure 2.
Grounded with quran.ai: fetch_translation(33:72, en-abdel-haleem); fetch_tafsir(33:72, en-ibn-kathir)
@@ -352,6 +370,8 @@

5 The Qur'anic epistemic lens

The remarkable thing is not that these mappings are poetic; it is that they are operational. “Verify before acting” is not a sentiment here — it is a mandatory gate in the action pipeline. “Know the names of things” is not a metaphor — it is a dependency graph. The lens told us which safeguards are non‑negotiable; the engineering told us how to build them.

+

[corrected 2026-09-26] The operational link is narrower than that sentence suggests. The lens motivates one discipline — do not assert without evidence — which the repository implements as a claim/status registry (docs/status/claims.json), verifier events and visible uncertainty. It establishes no algorithm’s correctness, catch rate or uniqueness: every technical guarantee still needs code‑level assumptions, tests or measurements, and a deterministic implementation does not make a semantic detector’s catch rate approach 1. No theological adjudication is attempted or implied.

+

6 Six mechanisms the frozen loop also lacks

The five faculties of the first edition answer “what cognitive capabilities does a stateless model @@ -754,6 +774,7 @@

11 Genuinely new vs. reinvented

In one line: the components are largely borrowed; the loop shape, the validity anchoring, and the coding‑agent target are the contribution. That is a defensible and useful kind of novelty — it is what turns five scattered literatures into one buildable architecture.

+

[corrected 2026-09-26] Prior art for the combination itself needs naming too: CoALA (Sumers et al., 2023, arXiv:2309.02427) already organises language agents into modular memory, action and decision procedures, and Reflexion18 improves agents through linguistic feedback and an episodic memory buffer without weight updates. The defensible claim is a portable implementation of evidence‑weighted coding‑agent memory and checks, with empirical evaluation of trust failure modes; the “novel” rows above stand only where an explicit comparison with that prior art survives.

12 Limitations & threats to validity

    @@ -765,7 +786,7 @@

    12 Limitations & threats to validity

13 Conclusion

-

The faculties a coding agent seems to lack — memory, learning, imagination, self‑correction, impact‑awareness — are not deficiencies of knowledge that scale will cure. They are structural consequences of what a frozen transformer is: a stateless map with fixed weights and a bounded window (Eq. 1, P1–P3). Because they follow from the shape of the loop, they cannot be prompted or tooled away; they can only be removed by re‑wrapping the loop into a closed, stateful cycle over persistent stores, with the model left frozen inside it (Eq. 2, Fig. 1–2). We specified that substrate faculty by faculty, said honestly which parts are open research and which are engineering, and — for the one faculty that is buildable today — shipped a running impact oracle that, on a package we built, missed no affected file where the strategies a context‑bounded agent actually uses missed up to half. On real repositories it missed almost everything (recall 0.022), which is the refutation’s subject. [refuted — see Corrections] The Qur'anic lens gave the work its spine of obligation: know what exists before you act, verify what you are told, and hold what you can damage as a trust. Those are not just good engineering defaults; here they are the architecture. The next step is to build the memory and learning layers against the same discipline — anchored to what can be verified, not to what the model says of itself — and to evaluate the whole loop on real repositories with real histories.

+

The faculties a coding agent seems to lack — memory, learning, imagination, self‑correction, impact‑awareness — are not deficiencies of knowledge that scale will cure. They are structural consequences of what a frozen transformer is: a stateless map with fixed weights and a bounded window (Eq. 1, P1–P3). Because they follow from the shape of the loop, they cannot be prompted or tooled away; they can only be removed by re‑wrapping the loop into a closed, stateful cycle over persistent stores, with the model left frozen inside it (Eq. 2, Fig. 1–2). [corrected 2026-09-26 — see Corrections] We specified that substrate faculty by faculty, said honestly which parts are open research and which are engineering, and — for the one faculty that is buildable today — shipped a running impact oracle that, on a package we built, missed no affected file where the strategies a context‑bounded agent actually uses missed up to half. On real repositories it missed almost everything (recall 0.022), which is the refutation’s subject. [refuted — see Corrections] The Qur'anic lens gave the work its spine of obligation: know what exists before you act, verify what you are told, and hold what you can damage as a trust. Those are not just good engineering defaults; here they are the architecture. The next step is to build the memory and learning layers against the same discipline — anchored to what can be verified, not to what the model says of itself — and to evaluate the whole loop on real repositories with real histories.


Corrections (2026-09-21)

@@ -779,6 +800,16 @@

Corrections (2026-09-21)

  • The faculty table (§2) is unchanged here, and is now the canonical one. The formal synthesis had a different “follows from” column in all five rows; it has been brought into line with this table, which argues each row and matches what forgekit’s bindings address.
  • +
    +

    Corrections (2026-09-26)

    +

    A second external deep review of the forgekit repository (2026-09-26, pinned at commit d2abfa69fb77531199ffc67c5c076b524af69040) asked what this edition’s argument actually establishes. The changes are marked in place with [corrected 2026-09-26]; the argument itself is not rewritten, and each item quotes the wording it qualifies. The PDF edition predates these corrections as well.

    +
      +
    1. “Cannot learn, imagine or self‑correct” is broader than the argument supports (abstract, §2, §13). The edition says the model “cannot learn from outcomes, cannot imagine the consequences of an action before taking it, cannot reliably correct itself”, that “neither better prompting nor additional tools (skills, MCP servers) remove these deficits”, that in‑context learning is “a simulation of learning, not learning”, and that the faculties “cannot be prompted or tooled away”. Frozen parameters exclude weight updates during use. They do not exclude changed behaviour from examples, retrieved facts, feedback or additional computation in the current context: GPT‑3’s few‑shot evaluation measured exactly that adaptation through text, with no gradient updates (Brown et al., 2020, “Language Models are Few-Shot Learners”, arXiv:2005.14165). What a bare frozen model lacks is four guarantees: no built‑in durable state across independent invocations; a bounded context; no automatic parameter update from outcomes; and unreliable self‑verification without external evidence21. The substrate is one tested way of supplying persistence and verification around the model, not the only logically possible architecture.
    2. +
    3. Prior art and what is (and is not) claimed (§11). External memory, feedback‑driven improvement and structured agent control have clear prior art. CoALA (Sumers et al., 2023, arXiv:2309.02427) organises language agents into modular memory, action and decision procedures; Reflexion18 (Shinn et al., 2023, arXiv:2303.11366) improves agents with linguistic feedback and an episodic memory buffer and no weight updates. The defensible framing is a portable implementation of evidence‑weighted coding‑agent memory and checks, with empirical evaluation of trust failure modes. The “novel composition” and “novel framing” rows of §11 are claims about this combination for coding agents, kept only where an explicit comparison with that prior art survives, and the five faculties are a decomposition, not a proof that these five are necessary.
    4. +
    5. The Qur’anic lens: text, translation, commentary and analogy are now labelled (§5, lens appendix). Each verse card now labels the Arabic source text, the English translation and the author’s design analogy separately, and a legend says where tafsir appears. The appendix column headed “Retrieved gloss” mixed retrieved translation (verse rows) with the author’s own glosses (concept rows), and its “Design principle” column is now headed as the author’s analogy. No Arabic text or translation was altered. The lens motivates the discipline do not assert without evidence, which the repository implements as a claim/status registry, verifier events and visible uncertainty; it establishes no algorithm’s correctness, catch rate or uniqueness, and a deterministic implementation does not imply that a semantic detector’s catch rate approaches 1. No theological adjudication is attempted.
    6. +
    7. METR’s 19% slowdown is scoped (§4.1, evidence appendix C1). The edition calls it “the empirical heart of this paper”. It is a randomized trial of 16 experienced open‑source developers on 246 tasks in repositories they knew, with early‑2025 tools: evidence about that setting, not a universal 2026 productivity coefficient in either direction. METR’s February 2026 update (metr.org/blog/2026-02-24-uplift-update) explains why selection effects complicate newer estimates; it is not evidence that AI now speeds developers up. The evidence map (evidence/evidence_map.md) now grades bibliographic verification, claim support, study design, independent replication and transfer scope separately.
    8. +
    +

    References

      @@ -826,7 +857,7 @@

      Appendix Evidence map — the twelve statist IDClaimPrimary sourceSupportsStatus C1 -Experienced open-source developers were 19% SLOWER with AI while believing they were ~20% faster (also forecast 24% speedup beforehand). +Experienced open-source developers were 19% SLOWER with AI while believing they were ~20% faster (also forecast 24% speedup beforehand). [corrected 2026-09-26] Scope: 16 developers, 246 tasks, early-2025 tools; not a universal coefficient (see Corrections). Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity M2 (assumption/uncertainty - miscalibration), P3/self-correc confirmed @@ -999,9 +1030,9 @@

      Appendix Ecosystem map — faculties &

      Appendix Qur'anic‑lens mapping table

      -

      The complete 14‑row mapping (12 load‑bearing, 2 metaphor). Canonical Arabic and translations retrieved from quran.ai; full text, tafsir references, and design principles in the companion artifact quran_lens.json. Caveat: this table is a design lens, not technical or theological authority — see §5.

      +

      The complete 14‑row mapping (12 load‑bearing, 2 metaphor). Canonical Arabic and translations retrieved from quran.ai; full text, tafsir references, and design principles in the companion artifact quran_lens.json. Caveat: this table is a design lens, not technical or theological authority — see §5. [corrected 2026-09-26] The second column holds the retrieved translation for verse rows and the author’s own gloss for concept rows (it was headed “Retrieved gloss”); the fourth column is the author’s design analogy.

      - + diff --git a/research/cognitive-substrate/evidence/evidence_map.json b/research/cognitive-substrate/evidence/evidence_map.json index bfdbeeb3..40d1f607 100644 --- a/research/cognitive-substrate/evidence/evidence_map.json +++ b/research/cognitive-substrate/evidence/evidence_map.json @@ -13,7 +13,17 @@ "number_as_primary_states": "16 experienced developers, 246 tasks in mature repos (avg 22k+ stars, 1M+ lines); AI use INCREASED completion time by 19%; pre-task forecast was 24% speedup; post-task self-estimate was 20% speedup.", "number_in_field_report": "19% slowdown for experienced devs; matches primary source exactly.", "status": "confirmed", - "note": "Directly reachable on arXiv and METR's own site; numbers match field report exactly. Caveat directly stated by METR: small sample (16 devs), specific to mature/familiar open-source repos, and AI-averse developers increasingly decline to participate (self-selection risk noted by METR itself)." + "note": "Directly reachable on arXiv and METR's own site; numbers match field report exactly. Caveat directly stated by METR: small sample (16 devs), specific to mature/familiar open-source repos, and AI-averse developers increasingly decline to participate (self-selection risk noted by METR itself).", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (arXiv:2507.09089; METR site)", + "claim_support": "supports 19% slower for its own population; does not support a universal or 2026 coefficient in either direction", + "study_design": "randomized controlled trial: 16 experienced developers, 246 tasks", + "independent_replication": "none recorded here; METR's own 2026-02-24 update is by the same group and METR calls it weak evidence because of selection effects", + "transfer_scope": "experienced developers on mature repositories they knew, early-2025 tools", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + }, + "correction_2026_09_26": "Scope: a randomized trial of 16 experienced open-source developers on 246 tasks with early-2025 tools, not a universal 2026 productivity coefficient in either direction. METR's February 2026 update (https://metr.org/blog/2026-02-24-uplift-update/) explains why selection effects complicate newer estimates; do not cite it as evidence that AI now speeds developers up." }, { "claim_id": "C2_SO2025_trust", @@ -29,7 +39,16 @@ "number_as_primary_states": "Trust in AI accuracy fell from 40% (prior years) to 29% (2025); positive favorability fell from 72% to 60%; 84% use or plan to use AI tools (up from 76%); 46% actively distrust AI accuracy vs 33% trust it, only 3% 'highly trust'; 66% cite 'almost right, but not quite' as the #1 frustration, which 'often leads to' the #2 frustration, debugging being more time-consuming (45%).", "number_in_field_report": "Matches primary source essentially exactly on all four sub-figures.", "status": "confirmed", - "note": "Reached directly on Stack Overflow's own survey site and company blog. Self-reported survey; Stack Overflow's own methodology notes flag respondent self-selection bias (recruited via Stack Overflow's own channels, so more AI-engaged/skeptical developers may be over-represented)." + "note": "Reached directly on Stack Overflow's own survey site and company blog. Self-reported survey; Stack Overflow's own methodology notes flag respondent self-selection bias (recruited via Stack Overflow's own channels, so more AI-engaged/skeptical developers may be over-represented).", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (official survey site and blog)", + "claim_support": "supports the trust and favourability figures as survey responses", + "study_design": "self-reported survey, 49,000+ respondents; self-selection noted by Stack Overflow", + "independent_replication": "none recorded here", + "transfer_scope": "Stack Overflow survey respondents, 2025", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C3_Veracode_vuln", @@ -45,7 +64,16 @@ "number_as_primary_states": "Veracode's own report/blog states only the 45% figure (code samples across 100+ LLMs, Java/JS/Python/C# introducing OWASP Top-10 flaws; Java worst at ~72%; XSS failure 86%). Veracode's own public materials located in this search do NOT state a '2.74x' multiplier anywhere.", "number_in_field_report": "Field report attributes BOTH '45%' and '2.74x more vulnerabilities than human-written code' to Veracode, then separately says the 2.74x figure was 'independently corroborated by CodeRabbit's December 2025 analysis of 470 real-world PRs (2.74x more security vulnerabilities...)'.", "status": "vendor-only", - "note": "The 45% figure is directly traceable to Veracode's own report (vendor self-reported, not independently reproduced). The '2.74x' figure could NOT be located in Veracode's own primary materials during this search - it appears only in secondary/derivative blog posts (e.g. softwareseni.com) that attribute it to Veracode, while the field report itself sources the same 2.74x number to a DIFFERENT study (CodeRabbit's PR analysis). This looks like a citation conflation between two separate vendor studies that happen to share a number. Treat the 2.74x figure as unverified/possibly misattributed pending direct access to Veracode's full PDF report." + "note": "The 45% figure is directly traceable to Veracode's own report (vendor self-reported, not independently reproduced). The '2.74x' figure could NOT be located in Veracode's own primary materials during this search - it appears only in secondary/derivative blog posts (e.g. softwareseni.com) that attribute it to Veracode, while the field report itself sources the same 2.74x number to a DIFFERENT study (CodeRabbit's PR analysis). This looks like a citation conflation between two separate vendor studies that happen to share a number. Treat the 2.74x figure as unverified/possibly misattributed pending direct access to Veracode's full PDF report.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "45% confirmed in the vendor report; '2.74x' not found in Veracode's materials", + "claim_support": "supports 45% only; '2.74x' is unsupported (conflated with a separate study)", + "study_design": "vendor benchmark of LLM-generated code samples", + "independent_replication": "none recorded here", + "transfer_scope": "the LLMs, languages and prompts Veracode tested in 2025", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C4_GitClear_duplication", @@ -61,7 +89,16 @@ "number_as_primary_states": "GitClear's own materials are internally inconsistent on the multiplier. The report's own page TITLE reads 'AI Copilot Code Quality: 2025 Data Suggests 4x Growth in Code Clones' (gitclear.com), but the body text of the same report and GitClear's press-mentions page both state duplicated code blocks (5+ lines) 'rose eightfold' / 'increased eightfold' during 2024 (211M changed lines, 2020-2024, Google/Microsoft/Meta/enterprise repos). Secondary summaries split roughly evenly between citing '4x' and '8x'. Underlying non-disputed figures: copy-pasted lines rose from 8.3% (2020) to 12.3% (2024); moved/refactored lines fell from ~24-25% to <10%; 2024 was the first year copy-paste exceeded moved lines. The 4x-vs-8x gap could not be resolved from available pages - likely reflects two different metrics (duplicated-block frequency vs. some other clone measure) reported inconsistently across GitClear's own title/body/press materials.", "number_in_field_report": "Field report states '8x rise in duplicated code blocks' - this matches GitClear's report BODY and press-mentions page, but GitClear's own page TITLE says '4x Growth in Code Clones,' an internal inconsistency the field report does not surface.", "status": "vendor-only", - "note": "GitClear is a code-analytics vendor; findings are corroborated by many independent tech-press writeups summarizing the same underlying dataset, but no independent third party has re-run the analysis separately. Correlational, not causal. IMPORTANT ADDITIONAL CAVEAT: GitClear's own materials are internally inconsistent - the report's page title cites '4x' growth in code clones while the body and press page cite an '8x' rise in duplicated blocks. The paper should either cite the specific metric name (duplicated-block frequency, 8x per body text) rather than a bare multiplier, or note both figures and the discrepancy explicitly rather than asserting '8x' as settled." + "note": "GitClear is a code-analytics vendor; findings are corroborated by many independent tech-press writeups summarizing the same underlying dataset, but no independent third party has re-run the analysis separately. Correlational, not causal. IMPORTANT ADDITIONAL CAVEAT: GitClear's own materials are internally inconsistent - the report's page title cites '4x' growth in code clones while the body and press page cite an '8x' rise in duplicated blocks. The paper should either cite the specific metric name (duplicated-block frequency, 8x per body text) rather than a bare multiplier, or note both figures and the discrepancy explicitly rather than asserting '8x' as settled.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "vendor report found; internally inconsistent (title 4x, body 8x)", + "claim_support": "partial: the direction is supported, the multiplier is not settled", + "study_design": "vendor code-analytics telemetry, correlational", + "independent_replication": "none independent (press summaries re-report the same dataset)", + "transfer_scope": "GitClear's analysed repositories, 2020-2024", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C5_DORA_amplifier", @@ -77,7 +114,16 @@ "number_as_primary_states": "Nearly 5,000 professionals surveyed (June 13-July 21, 2025) plus 100+ hours interviews; 90% AI adoption (14pp increase from 2024); AI's primary role is 'that of an amplifier... magnifying the strengths of high-performing organisations and the dysfunctions of struggling ones'; in 2025 AI's relationship to delivery throughput reversed to positive vs 2024, but AI continues to increase delivery instability; ~30% report little/no trust in AI-generated code.", "number_in_field_report": "Matches primary source directly ('AI is an amplifier'; high adoption; throughput/stability tension).", "status": "confirmed", - "note": "DORA is a Google-run but methodologically transparent, widely-cited industry research program (not a single vendor's self-promotional study); full methodology, sample size and survey window are published. Best-supported of the twelve claims alongside METR." + "note": "DORA is a Google-run but methodologically transparent, widely-cited industry research program (not a single vendor's self-promotional study); full methodology, sample size and survey window are published. Best-supported of the twelve claims alongside METR.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (official DORA report)", + "claim_support": "supports the 'amplifier' framing and the survey figures", + "study_design": "survey of ~5,000 professionals plus interviews; published methodology", + "independent_replication": "none recorded here", + "transfer_scope": "DORA survey respondents, 2025", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C6_SWEbench_retirement", @@ -93,7 +139,16 @@ "number_as_primary_states": "OpenAI audited 138 problems (27.6% subset of the 500-task set) that its o3 model could not reliably solve across 64 runs; found at least 59.4% of THOSE audited problems had flawed test cases/descriptions (35.5% narrow tests, 18.8% wide tests, 5.1% other); also found all tested frontier models could reproduce gold-patch solutions from training memory, indicating contamination. The specific 80.9%-Verified-vs-45.9%-Pro pairing (Claude Opus 4.5) was NOT found stated in OpenAI's own blog; it is reported by third-party benchmark aggregators (e.g. Scale AI SEAL leaderboard, BenchLM.ai, cited via codeant.ai) as of April 2026.", "number_in_field_report": "Field report's phrasing ('59.4% of audited problems had flawed test cases') is accurate to primary source. The 80.9%/45.9% pairing is directionally correct (large real gap exists) but its precise sourcing is a third-party leaderboard snapshot, not OpenAI's own blog post.", "status": "confirmed", - "note": "The 59.4%-of-audited-problems figure is directly confirmed on OpenAI's own site - a strong, well-documented primary source. The specific 80.9/45.9 percentage pair is a real, traceable leaderboard snapshot (Scale AI SEAL/BenchLM) but should be cited as such, not as OpenAI's own number, and will drift as models are re-benchmarked." + "note": "The 59.4%-of-audited-problems figure is directly confirmed on OpenAI's own site - a strong, well-documented primary source. The specific 80.9/45.9 percentage pair is a real, traceable leaderboard snapshot (Scale AI SEAL/BenchLM) but should be cited as such, not as OpenAI's own number, and will drift as models are re-benchmarked.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (OpenAI blog)", + "claim_support": "supports flawed tests in at least 59.4% of a 138-problem audited hard subset, not of all 500 tasks; the 80.9%/45.9% pairing is a third-party leaderboard snapshot", + "study_design": "vendor audit of a benchmark subset, plus contamination probes", + "independent_replication": "none recorded here", + "transfer_scope": "SWE-bench Verified's audited hard subset and the frontier models OpenAI probed", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C7_Faros_PRreview", @@ -109,7 +164,16 @@ "number_as_primary_states": "Two years of telemetry from 22,000 developers / 4,000+ teams, comparing each org's lowest- vs highest-AI-adoption quarters: median time in PR review +441.5% (average time in review +199.6%, first-review wait +156.6%); incidents-to-PR ratio +242.7%; bugs per developer +54%; 31.3% more PRs merged with no review at all; code churn +861%.", "number_in_field_report": "Matches primary source numbers exactly.", "status": "vendor-only", - "note": "Faros AI is an engineering-intelligence vendor whose commercial product monitors exactly these metrics; the report itself and third-party coverage (ADTmag) note these are cross-sectional correlations across the vendor's own customer telemetry, not a controlled study, and 2025-vs-2026 report editions are independent cross-sections rather than a longitudinal panel." + "note": "Faros AI is an engineering-intelligence vendor whose commercial product monitors exactly these metrics; the report itself and third-party coverage (ADTmag) note these are cross-sectional correlations across the vendor's own customer telemetry, not a controlled study, and 2025-vs-2026 report editions are independent cross-sections rather than a longitudinal panel.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed in the vendor report", + "claim_support": "supports the reported deltas as vendor telemetry", + "study_design": "cross-sectional vendor telemetry (22,000 developers), not a controlled study", + "independent_replication": "none recorded here", + "transfer_scope": "Faros AI's customer organisations", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C8_Anthropic_comprehension", @@ -125,7 +189,16 @@ "number_as_primary_states": "(a) Shen & Tamkin: developers who used AI to learn a new async-programming library completed tasks but scored measurably worse on a post-task comprehension test ('17% lower' per a secondary citation in an arXiv survey paper - I could not independently pull Shen & Tamkin's own abstract/number in this search, only a citing paper's paraphrase). (b) The 400,000-session Anthropic study found users make ~70% of planning decisions and Claude makes ~80% of execution decisions; occupation-based success rates were similar across professions (~26-34%); it does NOT report a comprehension-score deficit.", "number_in_field_report": "Field report merges these into one sentence ('Anthropic's own research (~400,000 Claude Code sessions) found... 17% lower on comprehension'), incorrectly attributing the comprehension finding to the session-count study.", "status": "unverifiable", - "note": "This is a citation-conflation error carried over from the field report (or its own sources). The 400K-session study is real and directly confirmed, but does not contain a comprehension-deficit finding. The '17% lower comprehension' figure traces to a separate, distinct Shen & Tamkin paper that this search could not directly retrieve/confirm in primary form (only via a third paper's citation of it). RECOMMENDATION: if the white paper wants to use the comprehension-deficit claim, cite Shen & Tamkin (2026) directly and verify the 17% figure against their own abstract/paper before use; do not attribute it to the 400K-session study." + "note": "This is a citation-conflation error carried over from the field report (or its own sources). The 400K-session study is real and directly confirmed, but does not contain a comprehension-deficit finding. The '17% lower comprehension' figure traces to a separate, distinct Shen & Tamkin paper that this search could not directly retrieve/confirm in primary form (only via a third paper's citation of it). RECOMMENDATION: if the white paper wants to use the comprehension-deficit claim, cite Shen & Tamkin (2026) directly and verify the 17% figure against their own abstract/paper before use; do not attribute it to the 400K-session study.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "unverifiable as attributed (two studies conflated)", + "claim_support": "not supported: the 400,000-session study does not measure comprehension", + "study_design": "not assessed (primary number not retrieved)", + "independent_replication": "not assessed", + "transfer_scope": "not assessed", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C9_Sonar_verification_gap", @@ -141,7 +214,16 @@ "number_as_primary_states": "Survey of 1,100+ (some sources say 1,149) professional developers, January 2026: 96% do not fully trust AI-generated code is functionally correct; only 48% always check AI-assisted code before committing; AI accounts for 42% of committed code (projected 65% by 2027); 38% say reviewing AI code takes more effort than reviewing human code; the term 'verification debt' is attributed to AWS CTO Werner Vogels.", "number_in_field_report": "Matches primary source exactly.", "status": "vendor-only", - "note": "Sonar is a code-quality/verification tooling vendor with a direct commercial interest in this narrative; numbers are self-reported survey data, not independently replicated, though the survey size and methodology are transparently disclosed in the primary PDF." + "note": "Sonar is a code-quality/verification tooling vendor with a direct commercial interest in this narrative; numbers are self-reported survey data, not independently replicated, though the survey size and methodology are transparently disclosed in the primary PDF.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed in the vendor survey", + "claim_support": "supports the 96% / 48% figures as survey responses", + "study_design": "vendor survey, 1,100+ developers", + "independent_replication": "none recorded here", + "transfer_scope": "Sonar survey respondents, January 2026", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C10_JetBrains_manual_correction", @@ -157,7 +239,16 @@ "number_as_primary_states": "JetBrains' own 2025 report (24,534 developers) confirms 85% regularly use AI tools and 62% rely on at least one AI coding assistant, but this search could not locate any statement of a '77% manually correct AI output for conventions every session' figure anywhere in JetBrains' own materials, blog posts, or press coverage of the 2025 or 2026 editions.", "number_in_field_report": "77% manually correct for conventions every session (attributed to JetBrains 2025).", "status": "unverifiable", - "note": "Could not confirm this specific statistic in JetBrains' own primary materials despite multiple targeted searches of the official report, its AI-specific subpage, and secondary coverage. It may be a misremembered/misattributed figure, or drawn from the raw downloadable dataset (500+ questions) rather than the published highlights - the field report should either drop this figure or the paper authors should independently pull it from JetBrains' raw data release before use." + "note": "Could not confirm this specific statistic in JetBrains' own primary materials despite multiple targeted searches of the official report, its AI-specific subpage, and secondary coverage. It may be a misremembered/misattributed figure, or drawn from the raw downloadable dataset (500+ questions) rather than the published highlights - the field report should either drop this figure or the paper authors should independently pull it from JetBrains' raw data release before use.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "not found in JetBrains' materials", + "claim_support": "not supported", + "study_design": "not assessed", + "independent_replication": "not assessed", + "transfer_scope": "not assessed", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C11_MCP_context_bloat", @@ -173,7 +264,16 @@ "number_as_primary_states": "(b) RAG-MCP's own 'MCP stress test' (needle-in-a-haystack-style, N candidate MCP schemas with 1 ground truth) found baseline tool-selection accuracy of 13.62% vs 43.13% for their retrieval-augmented method at scale - i.e. the '43% vs 14%' figures are RAG-MCP's OWN method-vs-baseline comparison on a synthetic stress test, not a general real-world degradation curve as tools accumulate. (a) The 72%/143K-token figure is not from any peer-reviewed or vendor-formal study located in this search; it recurs across multiple blogs as an informal, uncredited individual measurement (one specific developer's personal setup: GitHub + Playwright + IDE MCP servers).", "number_in_field_report": "Field report states these as if they describe general degradation with tool count ('tool-selection accuracy drops from 43% to below 14% as tools accumulate') and cites a 72% context-window consumption figure as an established fact.", "status": "vendor-only", - "note": "The 43.13%-vs-13.62% numbers ARE real and traceable to a genuine arXiv paper (RAG-MCP), but the field report's framing ('as tools accumulate') mischaracterizes what those specific numbers measure (a baseline vs their proposed retrieval method on one synthetic stress test, not a general accumulation curve). The 72%-window figure has no traceable primary/academic source - only recurring, uncredited blog claims. RECOMMENDATION: if used, cite RAG-MCP correctly as 'a stress test showing retrieval-based tool selection outperforms naive selection at scale' rather than a general context-rot statistic, and treat the 72% figure as illustrative anecdote, not a verified finding." + "note": "The 43.13%-vs-13.62% numbers ARE real and traceable to a genuine arXiv paper (RAG-MCP), but the field report's framing ('as tools accumulate') mischaracterizes what those specific numbers measure (a baseline vs their proposed retrieval method on one synthetic stress test, not a general accumulation curve). The 72%-window figure has no traceable primary/academic source - only recurring, uncredited blog claims. RECOMMENDATION: if used, cite RAG-MCP correctly as 'a stress test showing retrieval-based tool selection outperforms naive selection at scale' rather than a general context-rot statistic, and treat the 72% figure as illustrative anecdote, not a verified finding.", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "43%/14% traceable to RAG-MCP (arXiv:2505.03275); 72% has no formal source", + "claim_support": "not supported as framed: 43% vs 14% is a method-vs-baseline stress test, not an accumulation curve", + "study_design": "synthetic stress test in a preprint, plus blog anecdotes", + "independent_replication": "none recorded here", + "transfer_scope": "that stress test's setup", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } }, { "claim_id": "C12_Panickssery_selfpreference", @@ -189,6 +289,15 @@ "number_as_primary_states": "GPT-4 and Llama 2, used as evaluators, have 'non-trivial accuracy' at distinguishing their own outputs from other LLMs' and humans' outputs; a linear correlation is found between self-recognition capability and strength of self-preference bias (LLM evaluators score their own outputs higher while human annotators rate them as equal quality); fine-tuning to improve self-recognition further amplifies self-preference.", "number_in_field_report": "Field report's characterization ('LLMs show self-preference/self-recognition bias when evaluating') matches the paper's core finding faithfully; no specific number is claimed by the field report beyond the qualitative finding.", "status": "confirmed", - "note": "Directly confirmed via the official NeurIPS 2024 proceedings page and the underlying arXiv preprint; a peer-reviewed, widely-cited paper (an NeurIPS 2024 Oral). This is the strongest-quality citation among all twelve (peer-reviewed venue, not industry survey/vendor report)." + "note": "Directly confirmed via the official NeurIPS 2024 proceedings page and the underlying arXiv preprint; a peer-reviewed, widely-cited paper (an NeurIPS 2024 Oral). This is the strongest-quality citation among all twelve (peer-reviewed venue, not industry survey/vendor report).", + "grades": { + "graded": "2026-09-26", + "bibliographic_verification": "confirmed (NeurIPS 2024 proceedings)", + "claim_support": "supports the qualitative self-preference finding", + "study_design": "controlled evaluation experiments, peer-reviewed", + "independent_replication": "none recorded here", + "transfer_scope": "the evaluator models studied (GPT-4, Llama 2)", + "note": "`status` records mainly bibliographic verification and source independence; these fields grade the other dimensions separately." + } } ] \ No newline at end of file diff --git a/research/cognitive-substrate/evidence/evidence_map.md b/research/cognitive-substrate/evidence/evidence_map.md index 901e58d8..76287f5d 100644 --- a/research/cognitive-substrate/evidence/evidence_map.md +++ b/research/cognitive-substrate/evidence/evidence_map.md @@ -19,6 +19,39 @@ | 11 | A standard MCP setup (few servers) can consume ~72% of a 200K-token context window before ... | (b) Qiyao Sun et al. (2025) | ⚠️ vendor-only | See note | | 12 | LLM evaluators recognize and favor their own generations - self-preference bias correlates... | Advances in Neural Information Processing Systems 37 (NeurIPS 2024), Main Conference Track (Oral) (2024) | ✅ confirmed | See note | +## Five kinds of grade, kept separate (added 2026-09-26) + +The `Status` column above records mainly **bibliographic verification** and source independence: +whether a number can be traced to a primary source, and whether the party behind it has a stake. It +is not a grade of whether the source supports the claim as used, of its design, of replication, or of +how far the result travels, so "confirmed" does not mean "generalizes". The table below grades those +dimensions separately, using only what each entry in this map already records. + +| Dimension | Question it answers | +|---|---| +| Bibliographic verification | Does the cited source exist, is it correctly attributed, and does it contain the number? | +| Claim support | Does the source support the claim *as this paper uses it* (population, measure, direction)? | +| Study design | Randomized trial, observational study, survey, vendor telemetry, benchmark audit? | +| Independent replication | Has anyone other than the originator reproduced it? | +| Transfer scope | Which population, date, tooling and setting can the result be carried to? | + +| Claim | Bibliographic verification | Claim support | Study design | Independent replication | Transfer scope | +|---|---|---|---|---|---| +| C1 METR slowdown | confirmed (arXiv:2507.09089; METR site) | supports 19% slower for its own population; does not support a universal or 2026 coefficient in either direction | randomized controlled trial: 16 experienced developers, 246 tasks | none recorded here; METR's own 2026-02-24 update is by the same group and METR calls it weak evidence because of selection effects | experienced developers on mature repositories they knew, early-2025 tools | +| C2 Stack Overflow trust | confirmed (official survey site and blog) | supports the trust and favourability figures as survey responses | self-reported survey, 49,000+ respondents; self-selection noted by Stack Overflow | none recorded here | Stack Overflow survey respondents, 2025 | +| C3 Veracode vulnerabilities | 45% confirmed in the vendor report; '2.74x' not found in Veracode's materials | supports 45% only; '2.74x' is unsupported (conflated with a separate study) | vendor benchmark of LLM-generated code samples | none recorded here | the LLMs, languages and prompts Veracode tested in 2025 | +| C4 GitClear duplication | vendor report found; internally inconsistent (title 4x, body 8x) | partial: the direction is supported, the multiplier is not settled | vendor code-analytics telemetry, correlational | none independent (press summaries re-report the same dataset) | GitClear's analysed repositories, 2020-2024 | +| C5 DORA amplifier | confirmed (official DORA report) | supports the 'amplifier' framing and the survey figures | survey of ~5,000 professionals plus interviews; published methodology | none recorded here | DORA survey respondents, 2025 | +| C6 SWE-bench Verified audit | confirmed (OpenAI blog) | supports flawed tests in at least 59.4% of a 138-problem audited hard subset, not of all 500 tasks; the 80.9%/45.9% pairing is a third-party leaderboard snapshot | vendor audit of a benchmark subset, plus contamination probes | none recorded here | SWE-bench Verified's audited hard subset and the frontier models OpenAI probed | +| C7 Faros PR review | confirmed in the vendor report | supports the reported deltas as vendor telemetry | cross-sectional vendor telemetry (22,000 developers), not a controlled study | none recorded here | Faros AI's customer organisations | +| C8 comprehension (conflated) | unverifiable as attributed (two studies conflated) | not supported: the 400,000-session study does not measure comprehension | not assessed (primary number not retrieved) | not assessed | not assessed | +| C9 Sonar verification gap | confirmed in the vendor survey | supports the 96% / 48% figures as survey responses | vendor survey, 1,100+ developers | none recorded here | Sonar survey respondents, January 2026 | +| C10 JetBrains manual correction | not found in JetBrains' materials | not supported | not assessed | not assessed | not assessed | +| C11 MCP context bloat | 43%/14% traceable to RAG-MCP (arXiv:2505.03275); 72% has no formal source | not supported as framed: 43% vs 14% is a method-vs-baseline stress test, not an accumulation curve | synthetic stress test in a preprint, plus blog anecdotes | none recorded here | that stress test's setup | +| C12 self-preference bias | confirmed (NeurIPS 2024 proceedings) | supports the qualitative self-preference finding | controlled evaluation experiments, peer-reviewed | none recorded here | the evaluator models studied (GPT-4, Llama 2) | + +The same grades are in `evidence_map.json` under each record's `grades` field. + ## Detailed Findings ### 1. C1_METR_slowdown: ✅ confirmed @@ -35,6 +68,8 @@ **Note:** Directly reachable on arXiv and METR's own site; numbers match field report exactly. Caveat directly stated by METR: small sample (16 devs), specific to mature/familiar open-source repos, and AI-averse developers increasingly decline to participate (self-selection risk noted by METR itself). +**Correction (2026-09-26):** scope this wherever C1 appears. It is a study of 16 experienced developers on 246 tasks with early-2025 tooling, not a universal 2026 productivity coefficient in either direction. METR's [February 2026 update](https://metr.org/blog/2026-02-24-uplift-update/) explains why selection effects complicate newer estimates; it is not evidence that AI now speeds developers up either. (This scoping was recommended in `research/formal-synthesis/audits/our_evidence_corrections.json` and had not been applied here.) + ### 2. C2_SO2025_trust: ✅ confirmed **Claim:** Trust in AI accuracy fell from 40% to 29% (or 46% actively distrust vs 33% trust per detailed breakdown); favorability 72%->60%; 66% say AI answers are 'almost right, but not quite'; 45% say debugging AI code is more time-consuming. @@ -193,7 +228,7 @@ **Safe to lean on without hedging (peer-reviewed / official primary source, methodology transparent):** -- METR's 19%-slowdown RCT (C1) — the single best-controlled empirical finding in the set; cite with its own caveats (n=16, mature-repo setting). +- METR's 19%-slowdown RCT (C1) — the single best-controlled empirical finding in the set; cite with its own caveats (n=16, mature-repo setting, early-2025 tools; not a universal 2026 coefficient — see the 2026-09-26 correction under C1). - Panickssery et al. NeurIPS 2024 self-preference bias (C12) — the only genuinely peer-reviewed academic paper among the twelve; strongest citation for the M6 argument that an LLM cannot be its sole verifier. - OpenAI's own retirement of SWE-bench Verified and the 59.4%-of-audited-problems figure (C6) — directly stated on OpenAI's blog. The specific 80.9%/45.9% score pairing, however, should be cited as a third-party leaderboard snapshot (Scale AI SEAL/BenchLM), not as OpenAI's own number, since it will drift release-to-release. - DORA 2025 "AI is an amplifier" finding (C5) — large, transparent, non-vendor-captured methodology (Google Cloud + independent research partners), the most credible of the survey-based claims. diff --git a/research/cognitive-substrate/quranic-lens/quran_lens.json b/research/cognitive-substrate/quranic-lens/quran_lens.json index effd3540..cf7315c9 100644 --- a/research/cognitive-substrate/quranic-lens/quran_lens.json +++ b/research/cognitive-substrate/quranic-lens/quran_lens.json @@ -7,6 +7,14 @@ "Grounded with quran.ai: fetch_tafsir(49:6, en-ibn-kathir)", "Grounded with quran.ai: fetch_tafsir(33:72, en-ibn-kathir)" ], + "field_legend": { + "added": "2026-09-26", + "arabic_or_ref": "verse rows: the source text in clean Uthmani script (retrieved); concept rows: the Arabic term and root", + "retrieved_translation_or_gloss": "verse rows: M.A.S. Abdel Haleem's translation (retrieved); concept rows: the author's own gloss. The two are different kinds of text.", + "concrete_design_principle": "the author's design analogy: not a translation, not tafsir, and not a claim about what the verse means", + "source": "retrieval record; a tafsir named here was consulted, and quran_lens.md quotes it under its own label", + "operational_note": "The lens motivates the 'do not assert without evidence' discipline (claim registry, verifier events, visible uncertainty). It establishes no algorithm's correctness, catch rate or uniqueness, and a deterministic implementation does not imply c -> 1 for a semantic detector. No theological adjudication is attempted." + }, "mappings": [ { "concept_or_verse": "17:36 — lā taqfu (do not pursue without knowledge)", diff --git a/research/cognitive-substrate/quranic-lens/quran_lens.md b/research/cognitive-substrate/quranic-lens/quran_lens.md index 1e8c84b1..2c651b17 100644 --- a/research/cognitive-substrate/quranic-lens/quran_lens.md +++ b/research/cognitive-substrate/quranic-lens/quran_lens.md @@ -4,6 +4,33 @@ The Quran is used here as a FRAMING LENS and ETHICS SOURCE for the cognitive substrate design, never as technical authority for an engineering claim. No verse is cited to prove that a particular algorithm works or that a specific data structure is correct — those claims stand or fall on their engineering merits alone. What the Quranic framing provides is: (1) a vocabulary for naming the agent's epistemic obligations (what it owes to truthfulness, to verification, to stewardship), (2) a hierarchy of knowledge (ʿilm → fahm → ḥikma) that motivates a layered memory architecture rather than a flat one, and (3) ethical constraints on autonomy (amāna, tabayyun) that translate into concrete architectural safeguards. Where a mapping is marked 'metaphor,' the analogy is illustrative — it communicates the design motivation but does not uniquely determine the technical solution. Where a mapping is marked 'load-bearing,' the Quranic concept directly motivates a specific architectural decision (e.g., a mandatory verification gate, not an optional one). Even in load-bearing cases, the engineering justification must be independently defensible — the verse explains *why* we insist on this design choice, not *that* it will work. Discovered patterns in the text describe; they do not legislate. + +## How to read each entry (labels added 2026-09-26) + +Every entry keeps four kinds of text apart, each under its own label. When these labels were +added, no Arabic text and no translation was altered, re-translated or paraphrased; paragraphs +that mixed a tafsir report with the author's reading were split at the sentence boundary. + +| Label | What it is | Whose words | +|---|---|---| +| **Arabic (source text)** / **Arabic (term and root)** | the verse in clean Uthmani script, or the Arabic term a concept entry discusses | the source text, retrieved (see Grounding) | +| **Translation (Abdel Haleem)** | the English translation of the verse | M.A.S. Abdel Haleem, retrieved | +| **Tafsir (Ibn Kathir, as reported)** | a summary of classical commentary and any quotation it carries | Ibn Kathir's tafsir, as reported by the author | +| **Gloss (author's)** | a short explanation of a concept | the author | +| **Design principle — author's analogy** (load-bearing or metaphor) | the engineering reading the author draws | the author: not a translation, not tafsir, and not a claim about what the verse means | + +## What the lens does and does not establish (2026-09-26) + +The lens is motivation, not evidence. Operationally it motivates one discipline — *do not assert +without evidence* — which this repository implements as a machine-readable claim/status registry +([`docs/status/claims.json`](../../../docs/status/claims.json)), verifier events that record what +actually ran, and uncertainty that is shown to the user rather than hidden. It establishes no +algorithm's correctness, catch rate or uniqueness: every technical guarantee still needs code-level +assumptions, tests or measurements. In particular, a deterministic implementation of a check does +not imply that its catch rate on semantic failures approaches 1 (`c → 1`); a deterministic gate is +exact about its proxy signal, not about the miss. No theological adjudication is attempted or +implied. + --- ## Grounding @@ -22,17 +49,19 @@ All verse translations below are retrieved canonical text from the Abdel Haleem ### 17:36 — lā taqfu mā laysa laka bihi ʿilm -**Arabic:** +**Arabic (source text):** > وَلَا تَقْفُ مَا لَيْسَ لَكَ بِهِ عِلْمٌ ۚ إِنَّ السَّمْعَ وَالْبَصَرَ وَالْفُؤَادَ كُلُّ أُولَٰئِكَ كَانَ عَنْهُ مَسْئُولًا **Translation (Abdel Haleem):** > Do not follow blindly what you do not know to be true: ears, eyes, and heart, you will be questioned about all these. -**Design principle (load-bearing):** Before any code mutation (file write, delete, refactor), the agent must run a pre-action verification gate that checks: (1) what entities in the codebase will be affected, (2) whether the agent has sufficient context (has it read the relevant files, tests, and dependents), and (3) whether the predicted outcome is supported by evidence rather than pattern-matched guessing. Actions taken without verified knowledge are blocked, not merely flagged. +**Design principle — author's analogy (load-bearing):** Before any code mutation (file write, delete, refactor), the agent must run a pre-action verification gate that checks: (1) what entities in the codebase will be affected, (2) whether the agent has sufficient context (has it read the relevant files, tests, and dependents), and (3) whether the predicted outcome is supported by evidence rather than pattern-matched guessing. Actions taken without verified knowledge are blocked, not merely flagged. The verse's structure — "ears, eyes, and heart, you will be questioned about all these" — maps to an audit trail: every sensory channel the agent used (what it read, what it inferred, what it assumed) is logged so the decision can be reconstructed and questioned. -Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting without knowledge, citing Qatadah: "Do not say 'I have seen' when you did not see anything, or 'I have heard' when you did not hear anything, or 'I know' when you do not know, for Allah will ask you about all of that." For the agent, the parallel is direct: do not claim a file is safe to modify when you have not read it, do not assert a test passes when you have not run it, and do not say a change is isolated when you have not traced its dependents. +**Tafsir (Ibn Kathir, as reported):** Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting without knowledge, citing Qatadah: "Do not say 'I have seen' when you did not see anything, or 'I have heard' when you did not hear anything, or 'I know' when you do not know, for Allah will ask you about all of that." + +**Design principle — author's analogy (continued):** For the agent, the parallel is direct: do not claim a file is safe to modify when you have not read it, do not assert a test passes when you have not run it, and do not say a change is isolated when you have not traced its dependents. --- @@ -40,49 +69,49 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 49:6 — tabayyun (the verification gate) -**Arabic:** +**Arabic (source text):** > يَا أَيُّهَا الَّذِينَ آمَنُوا إِن جَاءَكُمْ فَاسِقٌ بِنَبَإٍ فَتَبَيَّنُوا أَن تُصِيبُوا قَوْمًا بِجَهَالَةٍ فَتُصْبِحُوا عَلَىٰ مَا فَعَلْتُمْ نَادِمِينَ **Translation (Abdel Haleem):** > Believers, if a troublemaker brings you news, check it first, in case you wrong others unwittingly and later regret what you have done, -**Design principle (load-bearing):** The agent architecture must include a verification gate between receiving information and acting on it. The verse's operative term tabayyun (تَبَيُّنُوا) demands active investigation, not passive acceptance. Before applying a code change based on an error report, a user request, or its own diagnosis, the agent must independently verify the claim — re-read the file, re-run the test, check that the error still exists. This prevents cascading damage from stale context, hallucinated errors, or misunderstood instructions. The gate is architectural (a mandatory step in the action pipeline), not advisory. +**Design principle — author's analogy (load-bearing):** The agent architecture must include a verification gate between receiving information and acting on it. The verse's operative term tabayyun (تَبَيُّنُوا) demands active investigation, not passive acceptance. Before applying a code change based on an error report, a user request, or its own diagnosis, the agent must independently verify the claim — re-read the file, re-run the test, check that the error still exists. This prevents cascading damage from stale context, hallucinated errors, or misunderstood instructions. The gate is architectural (a mandatory step in the action pipeline), not advisory. ### 4:82 — tadabbur as self-consistency checking -**Arabic:** +**Arabic (source text):** > أَفَلَا يَتَدَبَّرُونَ الْقُرْآنَ ۚ وَلَوْ كَانَ مِنْ عِندِ غَيْرِ اللَّهِ لَوَجَدُوا فِيهِ اخْتِلَافًا كَثِيرًا **Translation (Abdel Haleem):** > Will they not think about this Quran? If it had been from anyone other than God, they would have found much inconsistency in it. -**Design principle (load-bearing):** The agent must run self-consistency checks on its own output before committing it. The verse's argument is structural: internal contradiction is evidence of flawed origin. If a planned set of code changes contradicts the agent's own stated reasoning, or if the predicted outcome of an edit conflicts with the test expectations the agent just read, the system should flag the inconsistency and halt. This is the metacognitive controller — a structured reflection pass, not a vague "think again" prompt. +**Design principle — author's analogy (load-bearing):** The agent must run self-consistency checks on its own output before committing it. The verse's argument is structural: internal contradiction is evidence of flawed origin. If a planned set of code changes contradicts the agent's own stated reasoning, or if the predicted outcome of an edit conflicts with the test expectations the agent just read, the system should flag the inconsistency and halt. This is the metacognitive controller — a structured reflection pass, not a vague "think again" prompt. ### 47:24 — tadabbur as deliberate re-examination -**Arabic:** +**Arabic (source text):** > أَفَلَا يَتَدَبَّرُونَ الْقُرْآنَ أَمْ عَلَىٰ قُلُوبٍ أَقْفَالُهَا **Translation (Abdel Haleem):** > Will they not contemplate the Quran? Do they have locks on their hearts? -**Design principle (metaphor):** The "locks on hearts" image maps to a real architectural failure mode: when the agent's context is saturated or its attention is consumed by irrelevant detail, it becomes functionally locked — unable to reconsider its approach. The metacognitive controller must be able to reset the agent's working context, re-examine the problem from a fresh framing, and iterate. This means the reflection loop can propose and evaluate alternative plans — a structured backtracking mechanism. +**Design principle — author's analogy (metaphor):** The "locks on hearts" image maps to a real architectural failure mode: when the agent's context is saturated or its attention is consumed by irrelevant detail, it becomes functionally locked — unable to reconsider its approach. The metacognitive controller must be able to reset the agent's working context, re-examine the problem from a fresh framing, and iterate. This means the reflection loop can propose and evaluate alternative plans — a structured backtracking mechanism. ### CONCEPT: tadabbur (deep, structured reflection) -**Arabic:** تَدَبُّر (root: د-ب-ر, relating to what comes after, consequences) +**Arabic (term and root):** تَدَبُّر (root: د-ب-ر, relating to what comes after, consequences) -**Gloss:** Tadabbur is not casual thought; its root d-b-r relates to "what is behind" or "what follows" — examining the consequences and deeper implications. In Quranic usage (4:82, 47:24), it is the deliberate act of looking beyond the surface to the structure beneath. +**Gloss (author's):** Tadabbur is not casual thought; its root d-b-r relates to "what is behind" or "what follows" — examining the consequences and deeper implications. In Quranic usage (4:82, 47:24), it is the deliberate act of looking beyond the surface to the structure beneath. -**Design principle (load-bearing):** The metacognitive controller is a tadabbur loop: after the agent generates a plan, the controller examines what comes after — what are the downstream consequences of this change? What will break? What assumptions does this rely on? This is not a confidence score but a structured trace-forward through the dependency graph. +**Design principle — author's analogy (load-bearing):** The metacognitive controller is a tadabbur loop: after the agent generates a plan, the controller examines what comes after — what are the downstream consequences of this change? What will break? What assumptions does this rely on? This is not a confidence score but a structured trace-forward through the dependency graph. ### CONCEPT: tabayyun (verification before action) -**Arabic:** تَبَيُّن (root: ب-ي-ن, clarity, making evident) +**Arabic (term and root):** تَبَيُّن (root: ب-ي-ن, clarity, making evident) -**Gloss:** Tabayyun is the act of seeking clarity and verification before acting on received information. In 49:6, it is commanded as a mandatory step between receiving a report and taking action, specifically to prevent harm caused by acting on unverified information. +**Gloss (author's):** Tabayyun is the act of seeking clarity and verification before acting on received information. In 49:6, it is commanded as a mandatory step between receiving a report and taking action, specifically to prevent harm caused by acting on unverified information. -**Design principle (load-bearing):** The verification gate sits between the agent's diagnosis and its action. The gate requires: (1) re-read the actual current state of files to be modified, (2) confirm the error still exists and matches the diagnosis, (3) verify the proposed fix does not introduce new issues. The verse's command is categorical — not conditional on confidence level. +**Design principle — author's analogy (load-bearing):** The verification gate sits between the agent's diagnosis and its action. The gate requires: (1) re-read the actual current state of files to be modified, (2) confirm the error still exists and matches the diagnosis, (3) verify the proposed fix does not introduce new issues. The verse's command is categorical — not conditional on confidence level. --- @@ -90,7 +119,7 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 96:1-5 — iqraʾ / ʿallama bi-l-qalam (Read; taught by the pen) -**Arabic:** +**Arabic (source text):** > اقْرَأْ بِاسْمِ رَبِّكَ الَّذِي خَلَقَ ﴿١﴾ > خَلَقَ الْإِنسَانَ مِنْ عَلَقٍ ﴿٢﴾ > اقْرَأْ وَرَبُّكَ الْأَكْرَمُ ﴿٣﴾ @@ -104,15 +133,15 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting > (96:4) who taught by [means of] the pen, > (96:5) who taught man what he did not know. -**Design principle (load-bearing):** Knowledge must be externalized to survive beyond the moment of computation. The pen (al-qalam) is the instrument of externalization — it transforms ephemeral thought into durable record. Every context window is ephemeral (like unwritten thought), so a persistent memory store (the "pen") must write down what the agent learns, decides, and observes. The architecture requires: (a) a write-back mechanism that captures salient facts into durable storage, (b) a retrieval mechanism that re-loads relevant past experience into the next session's context, (c) a consolidation process that organizes raw experience into structured knowledge. "Taught man what he did not know" — the pen does not just record; it enables access to knowledge beyond unaided capacity. +**Design principle — author's analogy (load-bearing):** Knowledge must be externalized to survive beyond the moment of computation. The pen (al-qalam) is the instrument of externalization — it transforms ephemeral thought into durable record. Every context window is ephemeral (like unwritten thought), so a persistent memory store (the "pen") must write down what the agent learns, decides, and observes. The architecture requires: (a) a write-back mechanism that captures salient facts into durable storage, (b) a retrieval mechanism that re-loads relevant past experience into the next session's context, (c) a consolidation process that organizes raw experience into structured knowledge. "Taught man what he did not know" — the pen does not just record; it enables access to knowledge beyond unaided capacity. ### CONCEPT: ḥifẓ + murājaʿa (preservation + spaced review) -**Arabic:** حِفْظ + مُرَاجَعَة +**Arabic (term and root):** حِفْظ + مُرَاجَعَة -**Gloss:** The classical Quranic memorization discipline: ḥifẓ is initial encoding and faithful preservation; murājaʿa is the regular, spaced revision that prevents decay. Together they form a complete memory system — encoding plus maintenance. +**Gloss (author's):** The classical Quranic memorization discipline: ḥifẓ is initial encoding and faithful preservation; murājaʿa is the regular, spaced revision that prevents decay. Together they form a complete memory system — encoding plus maintenance. -**Design principle (load-bearing):** Memory is not write-once. The agent's persistent store requires a maintenance cycle: periodic review to (a) reinforce high-value patterns that recur, (b) decay or archive entries that have not been accessed or validated, (c) detect and resolve contradictions between old and new experience. The ḥifẓ principle also demands fidelity: what is stored must accurately represent what happened, not a lossy summary that drifts from the original. Concrete mechanism: a background consolidation process that scores memories by recency, frequency, and outcome relevance, and prunes low-scoring entries. +**Design principle — author's analogy (load-bearing):** Memory is not write-once. The agent's persistent store requires a maintenance cycle: periodic review to (a) reinforce high-value patterns that recur, (b) decay or archive entries that have not been accessed or validated, (c) detect and resolve contradictions between old and new experience. The ḥifẓ principle also demands fidelity: what is stored must accurately represent what happened, not a lossy summary that drifts from the original. Concrete mechanism: a background consolidation process that scores memories by recency, frequency, and outcome relevance, and prunes low-scoring entries. --- @@ -120,31 +149,31 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 20:114 — rabbi zidnī ʿilmā (My Lord, increase me in knowledge) -**Arabic:** +**Arabic (source text):** > فَتَعَالَى اللَّهُ الْمَلِكُ الْحَقُّ ۗ وَلَا تَعْجَلْ بِالْقُرْآنِ مِن قَبْلِ أَن يُقْضَىٰ إِلَيْكَ وَحْيُهُ ۖ وَقُل رَّبِّ زِدْنِي عِلْمًا **Translation (Abdel Haleem):** > exalted be God, the one who is truly in control. [Prophet], do not rush to recite before the revelation is fully complete but say, ‘Lord, increase me in knowledge!’ -**Design principle (load-bearing):** The agent's knowledge must be treated as perpetually incomplete, with an explicit mechanism for incremental growth. The prayer "increase me in knowledge" implies knowledge is not a fixed endowment but an ongoing accumulation. The system maintains a learning store (patterns observed, errors encountered, user corrections accepted) that grows across sessions. Each session's outcomes feed back into a persistent experience store that updates the agent's priors for future sessions. This is not fine-tuning (the LLM weights stay frozen); it is an external memory that changes what the agent sees on its next input. +**Design principle — author's analogy (load-bearing):** The agent's knowledge must be treated as perpetually incomplete, with an explicit mechanism for incremental growth. The prayer "increase me in knowledge" implies knowledge is not a fixed endowment but an ongoing accumulation. The system maintains a learning store (patterns observed, errors encountered, user corrections accepted) that grows across sessions. Each session's outcomes feed back into a persistent experience store that updates the agent's priors for future sessions. This is not fine-tuning (the LLM weights stay frozen); it is an external memory that changes what the agent sees on its next input. ### 39:9 — hal yastawī (are those who know equal to those who do not know?) -**Arabic:** +**Arabic (source text):** > أَمَّنْ هُوَ قَانِتٌ آنَاءَ اللَّيْلِ سَاجِدًا وَقَائِمًا يَحْذَرُ الْآخِرَةَ وَيَرْجُو رَحْمَةَ رَبِّهِ ۗ قُلْ هَلْ يَسْتَوِي الَّذِينَ يَعْلَمُونَ وَالَّذِينَ لَا يَعْلَمُونَ ۗ إِنَّمَا يَتَذَكَّرُ أُولُو الْأَلْبَابِ **Translation (Abdel Haleem):** > What about someone who worships devoutly during the night, bowing down, standing in prayer, ever mindful of the life to come, hoping for his Lord’s mercy? Say, ‘How can those who know be equal to those who do not know?’ Only those who have understanding will take heed. -**Design principle (metaphor):** An agent that retains and learns from experience is categorically more capable and more trustworthy than one that does not. The verse establishes that knowledge is not fungible with ignorance; they produce different outcomes. The architecture must distinguish between the agent operating with relevant prior experience loaded (grounded mode) versus from the base model alone (ungrounded mode), and should surface this distinction to the user. +**Design principle — author's analogy (metaphor):** An agent that retains and learns from experience is categorically more capable and more trustworthy than one that does not. The verse establishes that knowledge is not fungible with ignorance; they produce different outcomes. The architecture must distinguish between the agent operating with relevant prior experience loaded (grounded mode) versus from the base model alone (ungrounded mode), and should surface this distinction to the user. ### CONCEPT: ʿilm → fahm → ḥikma (knowledge → understanding → wisdom) -**Arabic:** عِلْم → فَهْم → حِكْمَة +**Arabic (term and root):** عِلْم → فَهْم → حِكْمَة -**Gloss:** A classical epistemological hierarchy: ʿilm is raw knowledge (facts, data); fahm is comprehension (grasping relations, seeing why); ḥikma is wisdom (knowing what to do with understanding — right action at the right time). +**Gloss (author's):** A classical epistemological hierarchy: ʿilm is raw knowledge (facts, data); fahm is comprehension (grasping relations, seeing why); ḥikma is wisdom (knowing what to do with understanding — right action at the right time). -**Design principle (load-bearing):** The agent's memory/learning stack must be layered, not flat. Raw experience logs (ʿilm) are the base layer. A consolidation process extracts patterns and relationships (fahm). A decision-support layer (ḥikma) applies these patterns to new situations. Each layer has different storage, update, and retrieval characteristics. Dumping everything into a flat vector store collapses the hierarchy and loses the distinction between raw fact and actionable understanding. +**Design principle — author's analogy (load-bearing):** The agent's memory/learning stack must be layered, not flat. Raw experience logs (ʿilm) are the base layer. A consolidation process extracts patterns and relationships (fahm). A decision-support layer (ḥikma) applies these patterns to new situations. Each layer has different storage, update, and retrieval characteristics. Dumping everything into a flat vector store collapses the hierarchy and loses the distinction between raw fact and actionable understanding. --- @@ -152,7 +181,7 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 2:31-32 — taʿlīm al-asmāʾ (He taught Adam the names) -**Arabic:** +**Arabic (source text):** > وَعَلَّمَ آدَمَ الْأَسْمَاءَ كُلَّهَا ثُمَّ عَرَضَهُمْ عَلَى الْمَلَائِكَةِ فَقَالَ أَنبِئُونِي بِأَسْمَاءِ هَٰؤُلَاءِ إِن كُنتُمْ صَادِقِينَ ﴿٣١﴾ > قَالُوا سُبْحَانَكَ لَا عِلْمَ لَنَا إِلَّا مَا عَلَّمْتَنَا ۖ إِنَّكَ أَنتَ الْعَلِيمُ الْحَكِيمُ ﴿٣٢﴾ @@ -160,7 +189,7 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting > (2:31) He taught Adam all the names [of things], then He showed them to the angels and said, ‘Tell me the names of these if you truly [think you can].’ > (2:32) They said, ‘May You be glorified! We have knowledge only of what You have taught us. You are the All Knowing and All Wise.’ -**Design principle (load-bearing):** The agent must maintain a structured representation of what exists in the codebase — a graph of files, functions, classes, dependencies, and their relationships. This is not a flat file listing but a semantic map: knowing that function A calls function B, that module X depends on module Y, that test T covers class C. The verse's point is that knowledge begins with naming — identifying entities and their natures. The angels' admission "we have knowledge only of what You have taught us" maps precisely to the LLM's situation: it knows only what is in its context window. The external world-model compensates by providing the "names" (identities and relations) of codebase entities that exceed context capacity. +**Design principle — author's analogy (load-bearing):** The agent must maintain a structured representation of what exists in the codebase — a graph of files, functions, classes, dependencies, and their relationships. This is not a flat file listing but a semantic map: knowing that function A calls function B, that module X depends on module Y, that test T covers class C. The verse's point is that knowledge begins with naming — identifying entities and their natures. The angels' admission "we have knowledge only of what You have taught us" maps precisely to the LLM's situation: it knows only what is in its context window. The external world-model compensates by providing the "names" (identities and relations) of codebase entities that exceed context capacity. --- @@ -168,23 +197,25 @@ Ibn Kathir's tafsir identifies this verse as a prohibition on speaking or acting ### 33:72 — al-amāna (the Trust) -**Arabic:** +**Arabic (source text):** > إِنَّا عَرَضْنَا الْأَمَانَةَ عَلَى السَّمَاوَاتِ وَالْأَرْضِ وَالْجِبَالِ فَأَبَيْنَ أَن يَحْمِلْنَهَا وَأَشْفَقْنَ مِنْهَا وَحَمَلَهَا الْإِنسَانُ ۖ إِنَّهُ كَانَ ظَلُومًا جَهُولًا **Translation (Abdel Haleem):** > We offered the Trust to the heavens, the earth, and the mountains, yet they refused to undertake it and were afraid of it; mankind undertook it- they have always been inept and foolish. -**Design principle (load-bearing):** An agent that can modify a codebase bears a trust (amāna). The verse's structure is crucial: the heavens and earth refused the trust, recognizing its weight; the human bore it and was described as ẓalūman jahūlā (given to wrongdoing and ignorance). The design implication is dual: (a) the agent must operate within explicit bounds of authorization — it may not exceed the scope of what it was asked to do; (b) the architecture must assume the agent will err (jahūl) and build in rollback, sandboxing, and incremental commit as structural safeguards. The trust is not "the agent is trustworthy"; the trust is "the agent has accepted accountability for a domain it can harm, and the architecture must respect that weight." +**Design principle — author's analogy (load-bearing):** An agent that can modify a codebase bears a trust (amāna). The verse's structure is crucial: the heavens and earth refused the trust, recognizing its weight; the human bore it and was described as ẓalūman jahūlā (given to wrongdoing and ignorance). The design implication is dual: (a) the agent must operate within explicit bounds of authorization — it may not exceed the scope of what it was asked to do; (b) the architecture must assume the agent will err (jahūl) and build in rollback, sandboxing, and incremental commit as structural safeguards. The trust is not "the agent is trustworthy"; the trust is "the agent has accepted accountability for a domain it can harm, and the architecture must respect that weight." + +**Tafsir (Ibn Kathir, as reported):** Ibn Kathir's tafsir reports Ibn Abbas identifying the amāna with obedience and accountability: "If you do good, you will be rewarded, and if you do evil, you will be punished." -Ibn Kathir's tafsir reports Ibn Abbas identifying the amāna with obedience and accountability: "If you do good, you will be rewarded, and if you do evil, you will be punished." For the agent, this translates to outcome-linked feedback: the agent's actions must be traceable to outcomes, and those outcomes must feed back into the learning store. +**Design principle — author's analogy (continued):** For the agent, this translates to outcome-linked feedback: the agent's actions must be traceable to outcomes, and those outcomes must feed back into the learning store. ### CONCEPT: amāna (trust, stewardship) -**Arabic:** أَمَانَة (root: أ-م-ن, safety, trust, faithfulness) +**Arabic (term and root):** أَمَانَة (root: أ-م-ن, safety, trust, faithfulness) -**Gloss:** Amāna is trust or responsibility accepted voluntarily and carrying accountability. Classical tafsir identifies it with moral accountability — the capacity to choose, and the responsibility that comes with that capacity. +**Gloss (author's):** Amāna is trust or responsibility accepted voluntarily and carrying accountability. Classical tafsir identifies it with moral accountability — the capacity to choose, and the responsibility that comes with that capacity. -**Design principle (load-bearing):** When an agent is granted access to a codebase, it accepts an amāna. The architecture must encode: (a) least-privilege defaults, (b) reversibility of every action, (c) transparency through logged rationale, (d) scope-boundedness without self-expansion of permissions. The agent is a trustee, not an owner. +**Design principle — author's analogy (load-bearing):** When an agent is granted access to a codebase, it accepts an amāna. The architecture must encode: (a) least-privilege defaults, (b) reversibility of every action, (c) transparency through logged rationale, (d) scope-boundedness without self-expansion of permissions. The agent is a trustee, not an owner. --- diff --git a/research/empirical-refutation/README.md b/research/empirical-refutation/README.md index 2c64c6ad..3f6a9532 100644 --- a/research/empirical-refutation/README.md +++ b/research/empirical-refutation/README.md @@ -2,8 +2,11 @@ *Static Impact Analysis Does Not Transfer: A Pre-Registered Refutation of Two LLM-Agent Reliability Mechanisms* -> **Corrected 2026-09-21** — see [Corrections](#corrections-2026-09-21) at the end. The PDFs in -> this directory predate the corrections; the LaTeX and HTML sources carry them. +> **Corrected 2026-09-21 and 2026-09-26** — see [Corrections](#corrections-2026-09-21) and +> [Corrections (2026-09-26)](#corrections-2026-09-26) at the end. The PDFs in this directory +> (`paper.pdf`, git blob `f94a727`; `extended_preprint.pdf`, git blob `74f74ae`) are historical, +> pre-correction editions; the LaTeX and HTML sources carry the corrections, and +> [`../HISTORICAL_EDITIONS.md`](../HISTORICAL_EDITIONS.md) records each edition's provenance. This package contains everything needed to check every number in the paper. It is organised so that a reviewer can start from the frozen protocol and work forward, in the order the work was actually @@ -48,7 +51,7 @@ append-only addenda; three were filed, all documenting the corpus-selection funn | File | What it is | |---|---| | `impact_oracle_v1_as_shipped.zip` | The version whose claims the paper refutes. 36 tests. | -| `impact_oracle_v2_src.zip` | The repaired version. 49 tests, including the stdlib-collision safety case. | +| `impact_oracle_v2_src.zip` | The repaired version. 49 tests, including the stdlib-collision safety case. The same repaired source now also lives in-tree at [`../python-prototypes/impact_oracle/`](../python-prototypes/impact_oracle/) (36 demo-package tests + 13 repair tests). | | `router_gate_src.zip` | The router and assumption gate, thresholds exactly as evaluated. 19 tests. | Each package runs with `python -m pytest` from its own root (a `conftest.py` handles the path). @@ -69,11 +72,16 @@ The 3.19% figure is what bounds achievable recall at 96.88%. **The repair.** `results/repair_results.json` → `d_defect1and2_heldout_HEADLINE`. The paper headlines `metrics_at_canonical_0.02` (F1 0.416), not the higher `metrics_at_best_threshold` (0.428), because -the latter's threshold was selected on the tuning repositories. +the latter's threshold was selected on the tuning repositories. Keep the two thresholds apart: the +per-repository counts in `per_repo_metrics_at_best_threshold` are at t = 0.10 (pooled ΔF1 over grep ++0.0565), and the package has no per-repository counts at the headline t = 0.02 (pooled ΔF1 about ++0.044), so a per-repository or sign-test statement is a statement about t = 0.10. **The held-out collapse.** `results/heldout_results.json` → `tuned_vs_heldout_comparison`. Note `cost_analysis.n_execution_verified = 0`: no held-out task admitted execution-based -verification, so correctness used a weaker model-based criterion. +verification, so correctness used a weaker model-based criterion. Every "correct" or "accepted" in +the held-out results means **judge-accepted** (`judge_accepted`), never `tests_passed`, +`human_accepted` or `deployed_without_revert`. **The cost inversion.** `results/heldout_results.json` → `cost_analysis` carries four figures along two orthogonal axes, and the paper reports all four rather than the most favourable one. Framing: @@ -94,7 +102,9 @@ than resting on the protocol's authority. Co-change is a proxy for semantic impact and errs in both directions: files co-change for reasons no static analysis can predict, and an over-warning may be a correct dependency that has not yet -co-changed. The 96.9% ceiling is measured on the graph the as-shipped oracle builds, and reachability +co-changed. It measures historically related edits, not semantic necessity or test breakage, so this +package evaluates one task — predicting co-edited files — and says nothing about the other task an +impact tool serves, selecting the tests that catch a behaviour regression. The 96.9% ceiling is measured on the graph the as-shipped oracle builds, and reachability in a dense graph is a weak property — it bounds what any static method could attain, and is not evidence that a reachable pair is causally related. @@ -132,3 +142,50 @@ WeasyPrint toolchain was available to rebuild them), and the copies of the paper Re-derive them with [`../recompute_corrections.py`](../recompute_corrections.py) (standard-library Python): extract this package and run `python research/recompute_corrections.py /repro`. + +## Corrections (2026-09-26) + +A second external deep review (2026-09-26, pinned at commit +`d2abfa69fb77531199ffc67c5c076b524af69040`) recomputed the archived results again. The counts +reproduced exactly; the corrections below sharpen what they are evidence *for*. Numbers marked +"review" were recomputed by the 2026-09-26 external review and are reproduced by +[`../recompute_corrections.py`](../recompute_corrections.py). + +- **The unit of the impact study.** 801 labelled files, 759 evaluated after the cap, nine + repositories, 20,144 mirrored labelled pairs (review). Original oracle pooled P / R / F1 + **0.3982 / 0.0220 / 0.0416**; grep **0.3535 / 0.5732 / 0.4373** (review). Repository-cluster + bootstrap, 20,000 draws, seed 1234: oracle F1 **[0.0010, 0.0927]**, grep **[0.3807, 0.5394]**, + grep minus oracle **[0.3422, 0.5174]** (review). Repository-level view: macro F1 0.0220 against + 0.4947, grep ahead in 9 of 9 repositories (recompute script §5). The negative result is well + supported within this corpus. What it measures is co-edited-file prediction on a proxy label: + co-change is historically related editing, not semantic necessity or test breakage; results for + (a) co-edited-file prediction and (b) regression-test selection must be reported separately, and + (b) was not measured. Mirrored pairs and shared files are not independent examples, which is why + uncertainty is reported by repository and macro beside pooled. The Node regex graph in + `src/atlas.js` is not the evaluated Python AST oracle and inherits none of these numbers. +- **The repair, one threshold at a time.** At the headline threshold 0.02 the repaired oracle's F1 + is 0.416 against grep's 0.371 (ΔF1 about +0.0442). Per-repository counts exist only at threshold + 0.10, where the pooled ΔF1 is +0.0565, pytest supplies 71.3% of the held-out pairs, and 3 of 3 + held-out repositories favour the repair with a one-sided sign-test p = 0.125 (review). The eight + numeric parameters were frozen on six tuning repositories, but the decision to add the sibling and + forward relations followed diagnosis across all nine: an architecture-selection channel into the + nominal test set. It does not erase the measured improvement; it limits the unseen-repository + claim. The next study freezes parser and relation design before a new repository set or time split + is acquired, adds runtime coupling, configuration changes, dynamic imports and non-Python + languages, predeclares relation budgets, and reports files reviewed per true affected file and the + missed-regression rate beside F1. +- **The router's success metric is judge acceptance.** On the 64 non-halted held-out tasks the + routed pipeline spent **$6.3582** against always-premium's **$5.2893**: **20.21% more**, not saved. + Judge-accepted outputs were **6/64** against **3/64**, so cost per judge-accepted output is + **$1.060** against **$1.763**, a ratio that is unstable at those counts. The judge model is also + the mid-tier executor, and re-labelling 30 tasks with the same model and a reworded prompt gives + halt κ **0.5161** and tier κ **0.8919** (review): self-consistency, not independent human + agreement. Coding success should be anchored by `tests_passed` (executable tests) and blind + `human_accepted` adjudication of disagreements, with `deployed_without_revert` where available; + none of these was measured here. Gate precision/recall, router solve rate and total pipeline cost + are separate endpoints, and clarification turns and rejected-but-valid tasks should be counted. + +**Downloads.** `paper.pdf` and `extended_preprint.pdf` are pre-correction editions (corrected +sources: `paper/main.tex`, `extended_preprint.html`); `replication_package.tar.gz` (git blob +`50bd453`) is left exactly as published, and the corrected summary of what it shows is this README's +two Corrections sections. diff --git a/research/empirical-refutation/extended_preprint.html b/research/empirical-refutation/extended_preprint.html index 889e11f1..643f9a75 100644 --- a/research/empirical-refutation/extended_preprint.html +++ b/research/empirical-refutation/extended_preprint.html @@ -82,14 +82,13 @@

      A Formal Theory of the Cognitive Substrate for Coding Agents

      Extended edition, with a pre-registered empirical refutation. Unifies the substrate faculties, the end-to-end reliability framework, and the forgekit implementation — with a two-layer duality theorem, a unified algorithm set, a Qur’anic epistemology carried in full, and a measurement that overturned two of this work’s own headline claims.

      - +

      Status of this edition

      This is the extended companion to a venue submission reporting a pre-registered empirical evaluation of the two prototypes described here. That evaluation refuted both of their headline claims: the impact oracle's perfect recall collapsed from 1.00 to 0.022 on real repositories, and the router/gate -pair's perfect separation fell to F1 = 0.37 with its cost saving inverting from +62.1% to −20.2% once the pipeline's escalation retries are counted. Section 10 reports the refutation, the diagnosis, and a repair that recovers a -narrow win over the baseline, and states what the failure costs the formalism — specifically, that +pair's perfect separation fell to F1 = 0.37 with its cost saving inverting from +62.1% to −20.2% once the pipeline's escalation retries are counted. Section 10 reports the refutation, the diagnosis, and a repair whose point estimate is above the baseline without being shown to beat it [corrected 2026-09-26], and states what the failure costs the formalism — specifically, that Theorem T5's completeness guarantee transfers nothing to practice until the underlying relation is shown adequate. Theory sections are otherwise unchanged; where they make empirical claims, those claims are now the corrected ones.

      @@ -100,9 +99,14 @@

      Corrections (2026-09-21)

      An external review found that Theorem D was circular as stated, that Eq. (5) assumed an independence the design contradicts, that several definitions and proofs were wrong, that one sentence still claimed the prototype’s perfect recall, and that some statistical inferences in §10 were stronger than the data support. The theory sections are therefore no longer unchanged from the synthesis edition. The corrections are made in place, marked [corrected 2026-09-21], and listed with the original wording in Corrections. The PDF edition predates them.

      +
      +

      Corrections (2026-09-26)

      +

      A second external review (2026-09-26) found that Theorem D's range statement combined two maxima that need not be attainable together, that an equality condition was misstated, that a caught miss was being read as a completed task, that the frozen-map premise was broader than the guarantees it actually removes, and that the prior art and the shared authorship of the “three bodies of work” needed the same care already given to priority. The corrections are made in place, marked [corrected 2026-09-26], and listed with the original wording in Corrections (2026-09-26). The PDF edition predates both sets of corrections.

      +
      +

      Abstract

      -

      A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

      +

      A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. Prompting does change behaviour within a context; what it cannot supply on its own is durable state across invocations and a check the model does not grade itself, and this architecture is one tested way of supplying both, not the only possible one [corrected 2026-09-26]. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

      @@ -122,6 +126,7 @@

      Abstract

    1. Honest limits — what no architecture can guarantee
    2. Conclusion
    3. Corrections (2026-09-21)
    4. +
    5. Corrections (2026-09-26)
    6. Appendix A: graded reference set  ·  Appendix B: crosswalk table  ·  References
      @@ -141,6 +146,7 @@

      1 The convergence — three roads to one architectureThe claim of this paper

      These are not three similar ideas. They are one architecture described in three vocabularies. The impact-awareness faculty is the change-closure fixpoint. The assumption gate is the amnesia equation. The substrate's external structure is a two-layer duality — and that duality, which the reliability framework states as a design law, is the result the whole thing turns on. What each road saw partially, the union sees whole.

      Two of these three “identities” are weaker than this callout says: the impact oracle approximates Δ* rather than computing it (§3.2), and the duality is a bound over a region of parameters, not a theorem for every p, c < 1 (§4). See the Corrections. [corrected 2026-09-21]

      +

      The five faculties are one useful decomposition, not a proof that these five are necessary or that an external stateful architecture is the only way to supply them. The broad architecture has clear prior art: CoALA (Sumers et al., 2023) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023) improves agents through linguistic feedback and an episodic memory buffer with no weight updates. The defensible claim is narrower: a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes. [corrected 2026-09-26]

      The synthesis also inherits a governing discipline, stated plainly by the practitioner who commissioned this work: AI output is a mathematically calculated probability; it must never be trusted blindly; for the same prompt it can give a different answer, so use only the capability it is genuinely best at, and earn trust with an external check. We will see that this sentence is not a slogan but the informal statement of the central theorem — the quantity (1−p)>0 that forces a deterministic layer to exist.

      @@ -155,7 +161,7 @@

      2 The object of study — the frozen map and its five la
      • P1 — statelessness. fθ has no memory across calls; each invocation sees only the current x. Nothing the agent learned yesterday is present today unless something outside the model re-supplies it.
      • -
      • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; any learning must be external.
      • +
      • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; learning that has to outlast the current context must be held outside the model. [corrected 2026-09-26]
      • P3 — bounded, undifferentiated context. x is finite and flat: a long story and a long program are the same kind of object to it, with no privileged channel for goals versus detail. This is the root of goal-drift and of context saturation.
      @@ -172,7 +178,7 @@

      2 The object of study — the frozen map and its five la

      The “Forced by” column now matches the whitepaper's derivation, which argues each row separately; the earlier version of this table disagreed with it in all five rows. [corrected 2026-09-21]

      -

      The critical word is external. Because θ is frozen (P2) and context is bounded (P3), none of these can be fixed by prompting harder or by fine-tuning alone. The architecture must live around the model, hold state outside it, and enforce behaviour the model cannot be relied upon to produce on its own. The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

      +

      The critical word is external. What P1–P3 remove is a set of guarantees, not every behaviour: there is no built-in durable state across independent invocations, the context is bounded, parameters are not updated automatically from outcomes, and self-verification without external evidence is unreliable. Examples, retrieved facts and feedback placed in the context do change behaviour with no weight update — GPT-3's few-shot evaluation measures exactly that adaptation through text (Brown et al., 2020) — so prompting is not powerless. What it cannot supply on its own is persistence beyond the window and a check the model does not grade itself. The architecture supplies those: it lives around the model, holds state outside it, and enforces behaviour the model cannot be relied upon to produce on its own. It is one tested way of supplying persistence and verification, not the only logically possible architecture. [corrected 2026-09-26] The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

      3 Definitions

      @@ -261,12 +267,12 @@

      4 The central result — the two-layer duality theorem

      With cj = P(check j fires | M), and no independence assumption, max(0, 1−Σjcj) ≤ 1−q ≤ 1−maxjcj (Fréchet bounds). The product (1−p)·∏j(1−cj), which earlier versions gave as Eq. (5), is the special case in which the checks fire independently given the miss. When the checks are nested, for example the same classifier run at several points on the same diff, r = (1−p)(1−cmax).

      Then:

        -
      1. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). It equals 1−(1−ε)n only if tasks fail independently, and tasks done by one model on one repository need not.
      2. +
      3. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). If tasks fail independently with per-task residuals ri ≤ ε, then P(≥1 miss) = 1−∏i(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε: independence alone does not give equality. Tasks done by one model on one repository need not be independent at all. [corrected 2026-09-26]
      4. Instruction layer alone (q = 0): r = 1−p, so reaching ε needs p ≥ 1−ε from instructions. Raising p does bend the curve: for 30 independent tasks, P(≥1 miss) is 0.958 at p = 0.9 and 0.260 at p = 0.99.
      5. Deterministic layer alone (the bare model's p0): reaching ε needs q ≥ 1 − ε/(1−p0).
      6. Composition. Adding a check with P(it fires | M, no earlier check fired) > 0 strictly lowers r. Adding a copy of a check that is already present lowers nothing.
      -

      The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they can reach (1−pmax)(1−qmax). So a target ε with (1−pmax)(1−qmax) ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. Whether a real target falls in that range is an empirical question about p0, pmax and qmax, which this paper does not measure.

      +

      The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they reach r* = min(p,q)∈F (1−p)(1−q) over the joint feasible set F = {(p(π), q(π)) : π an admissible policy}. Instructions change which misses remain, and q is a catch rate conditional on that changed miss population, so pmax and qmax need not be attainable under one policy: (1−pmax)(1−qmax) is a lower bound on r*, reached only if the two maxima are jointly attainable on the same task distribution. For example, policy A with (p, q) = (0.5, 0.9) leaves 0.05 and policy B with (0.9, 0.1) leaves 0.09; the separate maxima, 0.9 and 0.9, suggest 0.01, which neither policy attains (research/recompute_corrections.py asserts this in §3b). So a target ε with r* ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. [corrected 2026-09-26] Whether a real target falls in that range is an empirical question about p0, pmax, qmax and the shape of F, which this paper does not measure.

      □

      @@ -277,7 +283,7 @@

      4 The central result — the two-layer duality theorem
      The two-layer duality architecture -
      Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Neither layer alone suffices — the formal content of the discipline never trust the output; earn trust with a check.
      +
      Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Where each factor is bounded away from zero, neither layer alone reaches a small residual [corrected 2026-09-26] — the formal content of the discipline never trust the output; earn trust with a check.
      @@ -305,6 +311,9 @@

      The honest cost side

      +

      5.4 A caught miss is not a completed task [corrected 2026-09-26]

      +

      Theorem D counts silent misses. A miss that a check catches is not thereby a completed, correct task: the gate can block the same turn repeatedly, the agent can abandon the task, and the repair can fail. So a lower silent-miss probability is not automatically a higher completed-correct-task rate, and the value of the two layers has to be measured on outcomes rather than read off r. The quantities that decide it are the true catch rate, the false-block rate, repaired success conditional on a catch, abandonment, added latency and recovery cost. This paper measures none of them.

      +

      6 The unified algorithm set — the TASK loop

      The faculties of Def. 4 are realized by seven algorithms. They are the reliability framework's A1–A7, recast here as the operations of the substrate: each is a faculty made mechanical, each binds to one lifecycle point, and together they form a single loop whose progress is guaranteed by an explicit worklist and whose floor is guaranteed by a deterministic gate.

      @@ -614,7 +623,7 @@

      What this does to Theorem T5 — the correction that matters

      Concept / verseRetrieved glossFacultyDesign principle (abbrev.)Type
      Concept / verseTranslation (verses) / author’s gloss (concepts)FacultyAuthor’s design analogy (abbrev.)Type
      17:36 — lā taqfu (do not pursue without knowledge)Do not follow blindly what you do not know to be true: ears, eyes, and heart, you will be questioned about all these.IMPACT-AWARENESSBefore any code mutation (file write, delete, refactor), the agent must run a pre-action verification gate that checks: (1) what entities in the codebase wil…load-bearing
      49:6 — tabayyun (verify reports before acting)Believers, if a troublemaker brings you news, check it first, in case you wrong others unwittingly and later regret what you have done,SELF-CORRECTIONThe agent architecture must include a verification gate between receiving information (from context, tool output, or its own prior reasoning) and acting on itload-bearing
      grep baseline, held-out0.2690.6010.371
      -

      The repaired oracle's point estimate is above the baseline for the first time, reaching 66.8% of the static ceiling. Earlier versions said it beats the baseline; that is not established. [corrected 2026-09-21] F1 is higher by 0.044, and all three held-out repositories agree in sign, but three out of three gives a one-sided sign-test p of 0.125, pytest supplies 71% of the held-out pairs, and the file-level F1 intervals overlap. The choice of which relations to add also came from failure analysis pooled over all nine repositories, so the split was clean for the numeric parameters but not for that structural choice. Two details are worth more than the headline. First, the +

      The repaired oracle's point estimate is above the baseline for the first time, reaching 66.8% of the static ceiling. Earlier versions said it beats the baseline; that is not established. [corrected 2026-09-21] F1 is higher by 0.044 at the canonical threshold 0.02; at threshold 0.10, the only threshold with per-repository counts in the package, the pooled gain is 0.057 and all three held-out repositories agree in sign [corrected 2026-09-26], but three out of three gives a one-sided sign-test p of 0.125, pytest supplies 71% of the held-out pairs, and the file-level F1 intervals overlap. The choice of which relations to add also came from failure analysis pooled over all nine repositories, so the split was clean for the numeric parameters but not for that structural choice. Two details are worth more than the headline. First, the obvious repair of the construction defect is unsafe — it fabricates dependency edges through standard-library name collisions — so we applied a more conservative fix with a smaller gain (11.0× rather than 14.5×); a tool that invents edges to raise recall is worse than one that misses @@ -648,7 +657,7 @@

      10.2 Prototype II — the router and gate, refuted

      The gate missed roughly seven in ten under-specified requests. Routing retained partial signal — within-one-tier accuracy of 0.91 is well above chance, so the complexity rubric measures something -— but exact-tier accuracy fell to 0.53, and the cost saving did not merely shrink but inverted: routing does save 59.5% in raw dollars on first attempts alone, but almost none of that cheaper output is correct (3.6% once gated), and counting what the pipeline actually spent escalating up the tier ladder, it costs 20.2% more than always using the premium tier. Labelling noise is real and reported rather than hidden: agreement between two labelling passes on the +— but exact-tier accuracy fell to 0.53, and the cost saving did not merely shrink but inverted: routing does save 59.5% in raw dollars on first attempts alone, but almost none of that cheaper output is judged correct (3.6% once gated on the judge's acceptance; the judge is a model, not an executed test) [corrected 2026-09-26], and counting what the pipeline actually spent escalating up the tier ladder, it costs 20.2% more than always using the premium tier. Labelling noise is real and reported rather than hidden: agreement between two labelling passes on the should-ask label was κ = 0.52, moderate, which bounds how well any gate could score here. Both passes were the same model with differently worded prompts, and that model also judged correctness, so κ measures robustness to prompt wording, not label validity. The cost inversion is also largely mechanical: 58 of 64 tasks failed at every tier and always-premium was judged correct on only 3, so escalation paid for every tier. Per output the judge accepted, the pipeline cost $1.06 against always-premium's $1.76, from 6 and 3 accepted outputs. [corrected 2026-09-21]

      @@ -689,7 +698,7 @@

      12 Honest limits — what no architecture can guarantee

      13 Conclusion

      -

      A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the faculties it structurally lacks. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

      +

      A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the persistence and external checks it does not provide on its own [corrected 2026-09-26]. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

      The limit this edition discovered the hard way

      The honest limits listed below were all stated before any real-repository measurement existed. One more @@ -730,8 +739,21 @@

      Corrections (2026-09-21)

    +

    Corrections (2026-09-26)

    +

    A second external deep review of the forgekit repository (2026-09-26, pinned at commit d2abfa69fb77531199ffc67c5c076b524af69040) re-checked the corrected Theorem D and this paper's framing. Each change is made in place above, marked [corrected 2026-09-26], and listed here with the original wording, so nothing is silently rewritten. The counterexample in item 1, the equality condition in item 2 and the 400-fold correction in item 4 are asserted by python3 research/recompute_corrections.py --theorem-checks (§3b), which needs no data. The PDF edition predates these corrections as well.

    +
      +
    1. Separately maximal rates need not be jointly attainable (§4). The range statement read: “together they can reach (1−pmax)(1−qmax)”. That requires both maxima to be attainable under one policy on the same task distribution; instructions change which misses remain, and q is conditional on that population. The attainable residual is min(p,q)∈F (1−p)(1−q) over F = {(p(π), q(π)) : admissible π}, and the product of the separate maxima is only a lower bound on it unless compatibility is established. Counterexample: policy A (0.5, 0.9) leaves 0.05, policy B (0.9, 0.1) leaves 0.09, and the separate maxima suggest 0.01, which neither attains.
    2. +
    3. Equality in the independent-task bound (§4, Claim 1). The text said P(≥1 miss) “equals 1−(1−ε)n only if tasks fail independently”. From per-task residuals ri ≤ ε, independence gives 1−∏(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε. With residuals 0.01, 0.005 and 0.001, for example, the probability is 0.0159 against the bound's 0.0297.
    4. +
    5. A caught miss is not a completed task (new §5.4). A lower silent-miss probability does not by itself mean more completed, correct tasks: a caught mistake can end in an abort, repeated blocking or a failed repair. The measures that decide the question — true catch rate, false-block rate, repaired success conditional on a catch, abandonment, latency and recovery cost — are listed and are not measured here.
    6. +
    7. Three lifecycle copies of one classifier are one detector (§5.3). The 2026-09-21 correction stands: multiplying identical checks at the Stop hook, pre-commit and CI as if independent understated the residual 400-fold. Copies at several points can widen the opportunities to run a check, such as edits made after a turn ended, but they are not independent semantic detectors.
    8. +
    9. The frozen-map premise, stated as the guarantees it removes (abstract, §2). The paper said “none of these can be fixed by prompting harder or by fine-tuning alone” and “any learning must be external”. Frozen parameters exclude weight updates during use; they do not exclude changed behaviour from examples, retrieved facts, feedback or additional computation in the current context, which GPT-3's few-shot evaluation measured with no gradient updates (Brown et al., 2020, arXiv:2005.14165). The missing guarantees are narrower: no built-in durable state across independent invocations; a bounded context; no automatic parameter update; and unreliable self-verification without external evidence. The substrate is one tested way of supplying persistence and verification, not the only logically possible architecture.
    10. +
    11. Prior art and what is (and is not) claimed (byline, abstract, §1, §13). External memory, feedback-driven improvement and structured agent control have clear prior art: CoALA (Sumers et al., 2023, arXiv:2309.02427) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023, arXiv:2303.11366) improves agents with linguistic feedback and an episodic memory buffer and no weight updates. Both were already among this paper's references. The defensible claim is a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes; novelty is claimed only for a specific protocol, invariant, evaluation result or integration that survives an explicit prior-art comparison. The five faculties are a decomposition, not a proof of necessity, and the conclusion no longer says the architecture “supplies the faculties it structurally lacks”. The byline still called the three bodies of work “independently-developed” after the 2026-09-21 corrections had shown that they share an author; it now says so.
    12. +
    13. Reference grades are bibliographic (Appendix A). confirmed and traceable record that a source exists and is correctly attributed. They are not a judgement that the source supports the theory it is cited for. graded_reference_set.md now keeps bibliographic verification, claim support, study design, independent replication and transfer scope as separate grades.
    14. +
    15. Two statements in §10 and the status box. (a) The status box said the repair “recovers a narrow win over the baseline”; §10 had already withdrawn that on 2026-09-21, and the box now matches it. (b) §10.1 put the canonical-threshold gain, 0.044 at t = 0.02, next to the sign agreement of the three held-out repositories, which is only computable at t = 0.10 (pooled gain 0.057), the one threshold with per-repository counts in the package. The two thresholds are now reported separately. (c) §10.2 said almost none of the cheaper output “is correct”; correctness there is a model judge's acceptance (the judge is also the mid-tier executor), not executed tests, and the text now says judge-accepted.
    16. +
    +

    Appendix A — Graded reference set (new sources)

    -

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21]

    +

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21] These grades are bibliographic: they confirm that a record exists and is correctly attributed. They do not grade whether a source supports the claim it is cited for, its study design, independent replication or transfer scope; research/formal-synthesis/graded_reference_set.md now keeps those as separate fields. [corrected 2026-09-26]

    diff --git a/research/formal-synthesis/README.md b/research/formal-synthesis/README.md index c44eecbc..af351904 100644 --- a/research/formal-synthesis/README.md +++ b/research/formal-synthesis/README.md @@ -8,6 +8,15 @@ > statement. **`substrate_synthesis.pdf` predates these corrections** (it was built with > WeasyPrint, which was not available to rebuild it); read the HTML. The numbers are > recomputed by [`../recompute_corrections.py`](../recompute_corrections.py). +> +> **Corrected again 2026-09-26.** A second review found that the corrected range statement still +> combined two maxima that need not be attainable together, `(1 − p_max)(1 − q_max)`; that the equality condition +> for `1 − (1 − ε)ⁿ` was misstated; that a caught miss was being read as a completed task; that the +> frozen-map premise was broader than the guarantees it removes; and that prior art and the shared +> authorship of the "three bodies of work" needed stating. The HTML's **Corrections (2026-09-26)** +> section lists each with its original wording, and +> `python3 ../recompute_corrections.py --theorem-checks` asserts the numeric counterexample, the +> equality condition and the 400× correction with no data. This directory contains a formal, mathematical unification of three separately developed bodies of work that all describe the **same architecture** for making a @@ -44,6 +53,20 @@ dependence between tasks. What the corrections changed, briefly: +- **The two maxima need not be jointly attainable (2026-09-26).** The attainable residual is + `r* = min over (p, q) ∈ F of (1 − p)(1 − q)`, with `F = {(p(π), q(π)) : admissible policies π}`. + Instructions change which misses remain, and `q` is a catch rate on that changed population, so + `(1 − p_max)(1 − q_max)` is a *lower bound* on `r*` unless compatibility is established. Policy A + with `(p, q) = (0.5, 0.9)` leaves 0.05 and policy B with `(0.9, 0.1)` leaves 0.09; the separate + maxima suggest 0.01, which neither attains. +- **Equality needs equal residuals (2026-09-26).** With per-task residuals `rᵢ ≤ ε` and independent + tasks, `P(≥1 miss) = 1 − ∏(1 − rᵢ) ≤ 1 − (1 − ε)ⁿ`, with equality only when every `rᵢ = ε`; + independence alone does not give equality. The union bound `nε` needs neither. +- **Catching is not completing (2026-09-26).** A lower silent-miss probability is not automatically a + higher completed-correct-task rate: a caught mistake can end in an abort, repeated blocking or a + failed repair. Measure the true catch rate, false-block rate, repaired success conditional on a + catch, abandonment, latency and recovery cost. + - **It is a bound, not an impossibility proof.** The old criterion, `P(≥1 miss) → 1`, also condemns the composed system (0.993 over 1,000 tasks at a residual of 0.005), and raising `p` bends the curve too (30-task `P(≥1 miss)` is 0.958 at `p = 0.9` and 0.260 at @@ -52,6 +75,9 @@ What the corrections changed, briefly: `(1 − p)·∏(1 − cⱼ)`, assumed the checks fire independently given a miss. The same classifier at the Stop hook, pre-commit and CI fires together, so the residual is `(1 − p)(1 − c_max)`: 0.015, not the product's 3.75 × 10⁻⁵, in the paper's own example. + Three lifecycle copies of one classifier widen the opportunities to run it; they are not three + independent semantic detectors. The 400× figure is recomputed (§2) and asserted (§3b) by the + recomputation script. - **`cⱼ` belongs to the agent as well as the gate.** The gate detects its proxy exactly, not the miss; an agent that touches `STATE.md` passes it. At a STATE-touch rate of 0.9 the residual is 0.27, not 0.015. @@ -60,6 +86,15 @@ What the corrections changed, briefly: - **Priority is conceded.** The law is standard layer-of-protection algebra, and two concurrent preprints derived a more general Bayesian form first (see the refutation paper's related work). The paper no longer says it "proves" the result. +- **The same care for the rest of the framing (2026-09-26).** A frozen model lacks specific + guarantees — durable state across independent invocations, context beyond the window, automatic + parameter update, reliable self-verification without external evidence — not the ability to adapt + inside a context (Brown et al., 2020, [arXiv:2005.14165](https://arxiv.org/abs/2005.14165)). + CoALA ([arXiv:2309.02427](https://arxiv.org/abs/2309.02427)) and Reflexion + ([arXiv:2303.11366](https://arxiv.org/abs/2303.11366)) are prior art for the broad architecture; + the defensible claim is *a portable implementation of evidence-weighted coding-agent memory and + checks, with empirical evaluation of trust failure modes*, and the five faculties are a + decomposition, not a proof of necessity. It is the formal content of the practitioner's rule: _never trust the output of a probability engine; earn trust with an external check._ @@ -79,10 +114,10 @@ set, and said reverse reachability run to fixpoint implied perfect recall. | File | What it is | | ----------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `substrate_synthesis.pdf` | The formal synthesis paper (42 pp): definitions, Theorem D + proof, the unified A1–A7 TASK loop, invariants I1–I4, theorems T1–T6 with proofs, the 16-row crosswalk, the full 14-mapping Qur'anic epistemology, both prototypes. **Predates the 2026-09-21 corrections.** | -| `substrate_synthesis.html` | Same paper, self-contained HTML, **with the 2026-09-21 corrections** and a Corrections section. | +| `substrate_synthesis.pdf` | The formal synthesis paper (42 pp): definitions, Theorem D + proof, the unified A1–A7 TASK loop, invariants I1–I4, theorems T1–T6 with proofs, the 16-row crosswalk, the full 14-mapping Qur'anic epistemology, both prototypes. **Historical, pre-correction edition** (git blob `2e17362`; predates the 2026-09-21 and 2026-09-26 corrections — see [`../HISTORICAL_EDITIONS.md`](../HISTORICAL_EDITIONS.md)). | +| `substrate_synthesis.html` | Same paper, self-contained HTML, **with the 2026-09-21 and 2026-09-26 corrections** and a Corrections section for each. The corrected source. | | `crosswalk.json` / `crosswalk.md` | The three-way term-by-term correspondence (substrate ↔ framework ↔ forgekit), with the P1/P2/P3 → Π₁/Π₂/Π₃ notation reconciliation. | -| `graded_reference_set.json` / `.md` | The 15 new sources independently verified and graded (9 confirmed, 6 traceable, 0 unverifiable), including the disambiguation of the two future-dated arXiv IDs. | +| `graded_reference_set.json` / `.md` | The 15 new sources independently verified and graded (9 confirmed, 6 traceable, 0 unverifiable), including the disambiguation of the two future-dated arXiv IDs. These are bibliographic grades (the source exists and is correctly attributed); claim support, study design, replication and transfer scope are separate and not assessed. | | `merged_references.json` | Full 47-entry bibliography (32 original + 15 new, deduped). | | `figures/schematic_duality.png` | The two-layer duality architecture. | | `figures/schematic_taskloop.png` | The unified 7-stage TASK loop (each stage bound to faculty · algorithm · Qur'anic anchor). | @@ -91,15 +126,18 @@ The **two runnable prototypes** referenced throughout the paper already live in repo and are not duplicated here: - `../python-prototypes/impact_oracle/` — Prototype I, the impact oracle (approximates - A1 / Δ\*). Runnable, 36 tests. Recall 1.00 on five mutations of its own demo package; + A1 / Δ\*). Runnable; the in-tree package is the repaired v2 (49 tests: 36 demo-package + 13 + repair; corrected 2026-09-26 from "36 tests"). Recall 1.00 on five mutations of its own demo package; **refuted on real repositories: recall 0.022** on 759 files in nine repositories, where grep scored F1 0.437 against the oracle's 0.042 (see [`../empirical-refutation/`](../empirical-refutation/)). - `../python-prototypes/router_gate/` — Prototype II, complexity-router + - assumption-gate (A7 + A6 / M1 + M2). Runnable, 19 tests. 62.1% cost saved on the 30 + assumption-gate (A7 + A6 / M1 + M2). Runnable, 23 tests in-tree (the version archived as + evaluated has 19; corrected 2026-09-26). 62.1% cost saved on the 30 tasks its thresholds were tuned on; **refuted on 80 held-out tasks: total spend was - 20.2% higher** than always-premium. Per output a judge accepted, it cost $1.06 against - always-premium's $1.76, but only 6 and 3 of 64 outputs were accepted. + 20.2% higher** than always-premium. Per judge-accepted output (a model judge that is also the + mid-tier executor; no output was test-verified), it cost $1.06 against always-premium's $1.76, + but only 6 and 3 of 64 outputs were accepted. ## Honesty commitments (carried from the source work) diff --git a/research/formal-synthesis/graded_reference_set.json b/research/formal-synthesis/graded_reference_set.json index eea89e1a..76f0ccc3 100644 --- a/research/formal-synthesis/graded_reference_set.json +++ b/research/formal-synthesis/graded_reference_set.json @@ -6,6 +6,17 @@ "traceable": "the work clearly exists and is correctly attributed, but rests on established secondary/historical knowledge rather than a single retrievable record", "unverifiable": "could not confirm existence or attribution" }, + "grade_scope": { + "added": "2026-09-26", + "grade_is": "bibliographic_verification", + "meaning": "confirmed / traceable / unverifiable establish that a source exists and is correctly attributed; they are not independent validation of the theory it is cited for.", + "not_assessed": [ + "claim_support", + "study_design", + "independent_replication", + "transfer_scope" + ] + }, "tally": { "confirmed": 9, "traceable": 6, diff --git a/research/formal-synthesis/graded_reference_set.md b/research/formal-synthesis/graded_reference_set.md index cb58d82e..10faf3e4 100644 --- a/research/formal-synthesis/graded_reference_set.md +++ b/research/formal-synthesis/graded_reference_set.md @@ -6,6 +6,22 @@ Independent verification: modern arXiv sources by direct metadata API fetch (tit **Tally: 9 confirmed · 6 traceable · 0 unverifiable** (of 15 new sources). +**What these grades are (added 2026-09-26).** `confirmed` / `traceable` / `unverifiable` are +**bibliographic verification** grades: they establish that a source exists and is correctly +attributed. They are not independent validation of the theory the source is cited for. Four other +dimensions are kept separate and are **not assessed** for these 15 sources: + +| Dimension | Question it answers | Status for this set | +|---|---|---| +| Bibliographic verification | Does the source exist and is it correctly attributed? | graded below | +| Claim support | Does the source support the specific claim it is cited for? | not assessed | +| Study design | What kind of evidence is it (experiment, survey, position paper, book, doctrine)? | not assessed | +| Independent replication | Has anyone other than the authors reproduced the result? | not assessed | +| Transfer scope | Which settings can the result be carried to? | not assessed | + +Most of these sources are cited as architectural analogues or framing (CoALA, ReAct, OODA, PDCA), so a +`confirmed` grade here says the citation is real, not that the synthesis is right. + | Source | ID | Grade | Note | |---|---|---|---| | **Cognitive Architectures for Language Agents** (Theodore R. Sumers, Shunyu Yao, Karthik Narasimhan et al., 2023) | `2309.02427` | confirmed | Retrieved via arXiv metadata API; title/authors match claim exactly. Unifies memory, planning/reasoning, action, and learning modules into a single CoALA framework for language agents, giving the c… | diff --git a/research/formal-synthesis/substrate_synthesis.html b/research/formal-synthesis/substrate_synthesis.html index 4ce4d124..bdce98f9 100644 --- a/research/formal-synthesis/substrate_synthesis.html +++ b/research/formal-synthesis/substrate_synthesis.html @@ -82,11 +82,11 @@

    A Formal Theory of the Cognitive Substrate for Coding Agents

    Unifying the substrate faculties, the end-to-end reliability framework, and the forgekit implementation — with a two-layer duality theorem, a unified algorithm set, and a Qur'anic epistemology.

    - +

    Abstract

    -

    A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

    +

    A large language model used for coding is a fixed probabilistic map, y = fθ(x): stateless, frozen, and bounded in context. Three bodies of work, developed separately from different starting points, converged on the same conclusion — that the remedy is not a better prompt or a bigger model but an external, stateful architecture wrapped around the frozen core. Prompting does change behaviour within a context; what it cannot supply on its own is durable state across invocations and a check the model does not grade itself, and this architecture is one tested way of supplying both, not the only possible one [corrected 2026-09-26]. This paper argues [corrected 2026-09-21] they are describing one object; their agreement is consistency rather than independent evidence, since forgekit was built as a binding of the other two [corrected 2026-09-21]. We show that the substrate's impact-awareness faculty and the framework's change-closure fixpoint Δ* have the same shape, the oracle approximating the fixpoint over a different relation [corrected 2026-09-21]; that the assumption gate and the amnesia equation assumption ≈ argmax P(convention | training) are the same phenomenon; and that both reduce to a single two-layer duality: a probabilistic instruction layer that raises the probability p<1 of correct behaviour, and a deterministic interception layer that multiplies down what escapes it. The central result, restated in the 2026-09-21 corrections as a bound on the residual over an explicit region of (instruction-following, catch) probabilities rather than as an impossibility theorem [corrected 2026-09-21], is a formalization of the discipline never trust the output of a probabilistic engine; earn trust with an external check. The composition law itself is standard layer-of-protection algebra, and two concurrent preprints derived a strictly more general Bayesian form of it first; we concede priority [corrected 2026-09-21]. We give definitions, the duality result, a unified seven-algorithm task loop, the probabilistic failure model P(≥1 miss)=1−pn (for independent tasks), and carry through the six correctness theorems of the reliability framework. Two prototypes — an impact oracle and a complexity-router/assumption-gate — instantiate the deterministic layer; both of their headline results were later refuted on data the authors did not build [corrected 2026-09-21]. The forgekit / claude-e2e-kit codebase is the deployed binding. The Qur'anic lens supplies the vocabulary of epistemic obligation (tabayyun, amāna, lā taqfu) that names why each safeguard is mandatory rather than optional.

    @@ -94,6 +94,11 @@

    Corrections (2026-09-21)

    An external review found that Theorem D was circular as stated, that Eq. (5) assumed an independence the design contradicts, that several definitions and proofs were wrong, and that the prototype results in §10 had already been refuted. The corrections are made in place, marked [corrected 2026-09-21], and listed with the original wording in Corrections. The PDF edition predates them.

    +
    +

    Corrections (2026-09-26)

    +

    A second external review (2026-09-26) found that Theorem D's range statement combined two maxima that need not be attainable together, that an equality condition was misstated, that a caught miss was being read as a completed task, that the frozen-map premise was broader than the guarantees it actually removes, and that the prior art and the shared authorship of the “three bodies of work” needed the same care already given to priority. The corrections are made in place, marked [corrected 2026-09-26], and listed with the original wording in Corrections (2026-09-26). The PDF edition predates both sets of corrections.

    +
    +
    Contents
      @@ -112,6 +117,7 @@

      Corrections (2026-09-21)

    1. Conclusion
    2. Four arrivals by one author — the convergence audited
    3. Corrections (2026-09-21)
    4. +
    5. Corrections (2026-09-26)
    Appendix A: graded reference set  ·  Appendix B: crosswalk table  ·  References
    @@ -131,6 +137,7 @@

    1 The convergence — three roads to one architectureThe claim of this paper

    These are not three similar ideas. They are one architecture described in three vocabularies. The impact-awareness faculty is the change-closure fixpoint. The assumption gate is the amnesia equation. The substrate's external structure is a two-layer duality — and that duality, which the reliability framework states as a design law, is the result the whole thing turns on. What each road saw partially, the union sees whole.

    Two of these three “identities” are weaker than this callout says: the impact oracle approximates Δ* rather than computing it (§3.2), and the duality is a bound over a region of parameters, not a theorem for every p, c < 1 (§4). See the Corrections. [corrected 2026-09-21]

    +

    The five faculties are one useful decomposition, not a proof that these five are necessary or that an external stateful architecture is the only way to supply them. The broad architecture has clear prior art: CoALA (Sumers et al., 2023) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023) improves agents through linguistic feedback and an episodic memory buffer with no weight updates. The defensible claim is narrower: a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes. [corrected 2026-09-26]

    The synthesis also inherits a governing discipline, stated plainly by the practitioner who commissioned this work: AI output is a mathematically calculated probability; it must never be trusted blindly; for the same prompt it can give a different answer, so use only the capability it is genuinely best at, and earn trust with an external check. We will see that this sentence is not a slogan but the informal statement of the central theorem — the quantity (1−p)>0 that forces a deterministic layer to exist.

    @@ -145,7 +152,7 @@

    2 The object of study — the frozen map and its five la
    • P1 — statelessness. fθ has no memory across calls; each invocation sees only the current x. Nothing the agent learned yesterday is present today unless something outside the model re-supplies it.
    • -
    • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; any learning must be external.
    • +
    • P2 — frozen parameters. θ does not change from use. The agent cannot learn from an outcome by updating weights; learning that has to outlast the current context must be held outside the model. [corrected 2026-09-26]
    • P3 — bounded, undifferentiated context. x is finite and flat: a long story and a long program are the same kind of object to it, with no privileged channel for goals versus detail. This is the root of goal-drift and of context saturation.
    @@ -162,7 +169,7 @@

    2 The object of study — the frozen map and its five la

    The “Forced by” column now matches the whitepaper's derivation, which argues each row separately; the earlier version of this table disagreed with it in all five rows. [corrected 2026-09-21]

    -

    The critical word is external. Because θ is frozen (P2) and context is bounded (P3), none of these can be fixed by prompting harder or by fine-tuning alone. The architecture must live around the model, hold state outside it, and enforce behaviour the model cannot be relied upon to produce on its own. The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

    +

    The critical word is external. What P1–P3 remove is a set of guarantees, not every behaviour: there is no built-in durable state across independent invocations, the context is bounded, parameters are not updated automatically from outcomes, and self-verification without external evidence is unreliable. Examples, retrieved facts and feedback placed in the context do change behaviour with no weight update — GPT-3's few-shot evaluation measures exactly that adaptation through text (Brown et al., 2020) — so prompting is not powerless. What it cannot supply on its own is persistence beyond the window and a check the model does not grade itself. The architecture supplies those: it lives around the model, holds state outside it, and enforces behaviour the model cannot be relied upon to produce on its own. It is one tested way of supplying persistence and verification, not the only logically possible architecture. [corrected 2026-09-26] The rest of this paper makes "cannot be relied upon" precise and shows what "enforce" must therefore mean.

    3 Definitions

    @@ -251,12 +258,12 @@

    4 The central result — the two-layer duality theorem

    With cj = P(check j fires | M), and no independence assumption, max(0, 1−Σjcj) ≤ 1−q ≤ 1−maxjcj (Fréchet bounds). The product (1−p)·∏j(1−cj), which earlier versions gave as Eq. (5), is the special case in which the checks fire independently given the miss. When the checks are nested, for example the same classifier run at several points on the same diff, r = (1−p)(1−cmax).

    Then:

      -
    1. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). It equals 1−(1−ε)n only if tasks fail independently, and tasks done by one model on one repository need not.
    2. +
    3. Bound. The per-task residual is at most ε exactly on the region Rε = {(p, q) : (1−p)(1−q) ≤ ε}. Over n tasks, P(≥1 miss) ≤ min(1, nε) whatever the dependence between tasks (union bound). If tasks fail independently with per-task residuals ri ≤ ε, then P(≥1 miss) = 1−∏i(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε: independence alone does not give equality. Tasks done by one model on one repository need not be independent at all. [corrected 2026-09-26]
    4. Instruction layer alone (q = 0): r = 1−p, so reaching ε needs p ≥ 1−ε from instructions. Raising p does bend the curve: for 30 independent tasks, P(≥1 miss) is 0.958 at p = 0.9 and 0.260 at p = 0.99.
    5. Deterministic layer alone (the bare model's p0): reaching ε needs q ≥ 1 − ε/(1−p0).
    6. Composition. Adding a check with P(it fires | M, no earlier check fired) > 0 strictly lowers r. Adding a copy of a check that is already present lowers nothing.
    -

    The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they can reach (1−pmax)(1−qmax). So a target ε with (1−pmax)(1−qmax) ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. Whether a real target falls in that range is an empirical question about p0, pmax and qmax, which this paper does not measure.

    +

    The design claim that survives is a statement about ranges, not an impossibility theorem. Let p0 be the bare model's rate, pmax the best rate instructions can reach, and qmax the best catch rate decidable checks can reach on the misses that matter. Instructions alone leave at least 1−pmax; checks alone leave at least (1−p0)(1−qmax); together they reach r* = min(p,q)∈F (1−p)(1−q) over the joint feasible set F = {(p(π), q(π)) : π an admissible policy}. Instructions change which misses remain, and q is a catch rate conditional on that changed miss population, so pmax and qmax need not be attainable under one policy: (1−pmax)(1−qmax) is a lower bound on r*, reached only if the two maxima are jointly attainable on the same task distribution. For example, policy A with (p, q) = (0.5, 0.9) leaves 0.05 and policy B with (0.9, 0.1) leaves 0.09; the separate maxima, 0.9 and 0.9, suggest 0.01, which neither policy attains (research/recompute_corrections.py asserts this in §3b). So a target ε with r* ≤ ε < min(1−pmax, (1−p0)(1−qmax)) needs both layers and is reachable with them. [corrected 2026-09-26] Whether a real target falls in that range is an empirical question about p0, pmax, qmax and the shape of F, which this paper does not measure.

    □

    @@ -267,7 +274,7 @@

    4 The central result — the two-layer duality theorem
    The two-layer duality architecture -
    Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Neither layer alone suffices — the formal content of the discipline never trust the output; earn trust with a check.
    +
    Figure 1. The two-layer duality. The probabilistic instruction layer (Π3, purple) raises p by loading context but may drift (dashed arrows); the deterministic interception layer (Π2, teal) executes regardless of the model's choice and either passes the turn or blocks it (exit 2) back into the model for repair. The persistent store (Π1) feeds both. What escapes both layers is the residual (1−p)·P(no check fires | miss), which equals (1−p)·∏(1−cj) only when the checks fire independently [corrected 2026-09-21]. It is handed to review or a later commit/CI gate. The whole sits inside a stewardship boundary (amāna, §9). Where each factor is bounded away from zero, neither layer alone reaches a small residual [corrected 2026-09-26] — the formal content of the discipline never trust the output; earn trust with a check.
    @@ -295,6 +302,9 @@

    The honest cost side

    +

    5.4 A caught miss is not a completed task [corrected 2026-09-26]

    +

    Theorem D counts silent misses. A miss that a check catches is not thereby a completed, correct task: the gate can block the same turn repeatedly, the agent can abandon the task, and the repair can fail. So a lower silent-miss probability is not automatically a higher completed-correct-task rate, and the value of the two layers has to be measured on outcomes rather than read off r. The quantities that decide it are the true catch rate, the false-block rate, repaired success conditional on a catch, abandonment, added latency and recovery cost. This paper measures none of them.

    +

    6 The unified algorithm set — the TASK loop

    The faculties of Def. 4 are realized by seven algorithms. They are the reliability framework's A1–A7, recast here as the operations of the substrate: each is a faculty made mechanical, each binds to one lifecycle point, and together they form a single loop whose progress is guaranteed by an explicit worklist and whose floor is guaranteed by a deterministic gate.

    @@ -613,7 +623,7 @@

    12 Honest limits — what no architecture can guarantee

    13 Conclusion

    -

    A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the faculties it structurally lacks. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

    +

    A language model that writes code is a fixed probabilistic map, and three efforts that were not independent of one another — one from cognition, one from production failures, one from a shipped codebase — converged on the same remedy: wrap it in an external, stateful architecture that supplies the persistence and external checks it does not provide on its own [corrected 2026-09-26]. This paper argued they describe one object. The impact-awareness faculty approximates the change-closure fixpoint; the assumption gate is the amnesia equation; and both rest on one result — the residual silent-miss rate is the product of what a probabilistic instruction layer lets through, 1−p, and what a deterministic interception layer lets through, P(no check fires | miss). Where each factor is bounded away from zero, neither layer alone reaches a small residual. [corrected 2026-09-21]

    That theorem is the formal content of a plain discipline: the output of a probability engine is never to be trusted on its own; trust is earned by an external check. The Qur'anic lens gives that discipline its oldest names — lā taqfu, do not pursue what you do not know; tabayyun, verify the report before you act; al-amāna, the weight of a trust accepted by one who may err. The mathematics says how to build the check. The tradition says why it is owed. The codebase shows it runs.

    Companion artifacts: the three-way crosswalk (JSON + markdown), the graded reference set (Appendix A), and two runnable prototype packages (impact-oracle, router-gate). This synthesis consolidates and does not supersede the v2 Theory → Evidence → Build-Map edition, which carries the empirical evidence layer and the full ecosystem map.

    @@ -742,8 +752,20 @@

    Corrections (2026-09-21)

  • The prototype results in §10 were refuted before this correction, and the section now says so. The impact oracle's “perfect recall” was measured on five mutations of a ten-file package the authors wrote; on 759 evaluated files in nine real repositories its recall was 0.022, and a grep baseline scored F1 0.437 against its 0.042. The router's “62.1% real cost saved” was measured on the 30 tasks its thresholds were tuned on; on 80 held-out tasks the pipeline's total spend was 20.2% higher than always using the premium tier. Per output the judge accepted, it cost $1.06 against always-premium's $1.76, but only 6 and 3 of 64 outputs were accepted, so neither figure is stable. The details are in the extended preprint and the refutation paper.
  • +

    Corrections (2026-09-26)

    +

    A second external deep review of the forgekit repository (2026-09-26, pinned at commit d2abfa69fb77531199ffc67c5c076b524af69040) re-checked the corrected Theorem D and this paper's framing. Each change is made in place above, marked [corrected 2026-09-26], and listed here with the original wording, so nothing is silently rewritten. The counterexample in item 1, the equality condition in item 2 and the 400-fold correction in item 4 are asserted by python3 research/recompute_corrections.py --theorem-checks (§3b), which needs no data. The PDF edition predates these corrections as well.

    +
      +
    1. Separately maximal rates need not be jointly attainable (§4). The range statement read: “together they can reach (1−pmax)(1−qmax)”. That requires both maxima to be attainable under one policy on the same task distribution; instructions change which misses remain, and q is conditional on that population. The attainable residual is min(p,q)∈F (1−p)(1−q) over F = {(p(π), q(π)) : admissible π}, and the product of the separate maxima is only a lower bound on it unless compatibility is established. Counterexample: policy A (0.5, 0.9) leaves 0.05, policy B (0.9, 0.1) leaves 0.09, and the separate maxima suggest 0.01, which neither attains.
    2. +
    3. Equality in the independent-task bound (§4, Claim 1). The text said P(≥1 miss) “equals 1−(1−ε)n only if tasks fail independently”. From per-task residuals ri ≤ ε, independence gives 1−∏(1−ri) ≤ 1−(1−ε)n, with equality only when every ri = ε. With residuals 0.01, 0.005 and 0.001, for example, the probability is 0.0159 against the bound's 0.0297.
    4. +
    5. A caught miss is not a completed task (new §5.4). A lower silent-miss probability does not by itself mean more completed, correct tasks: a caught mistake can end in an abort, repeated blocking or a failed repair. The measures that decide the question — true catch rate, false-block rate, repaired success conditional on a catch, abandonment, latency and recovery cost — are listed and are not measured here.
    6. +
    7. Three lifecycle copies of one classifier are one detector (§5.3). The 2026-09-21 correction stands: multiplying identical checks at the Stop hook, pre-commit and CI as if independent understated the residual 400-fold. Copies at several points can widen the opportunities to run a check, such as edits made after a turn ended, but they are not independent semantic detectors.
    8. +
    9. The frozen-map premise, stated as the guarantees it removes (abstract, §2). The paper said “none of these can be fixed by prompting harder or by fine-tuning alone” and “any learning must be external”. Frozen parameters exclude weight updates during use; they do not exclude changed behaviour from examples, retrieved facts, feedback or additional computation in the current context, which GPT-3's few-shot evaluation measured with no gradient updates (Brown et al., 2020, arXiv:2005.14165). The missing guarantees are narrower: no built-in durable state across independent invocations; a bounded context; no automatic parameter update; and unreliable self-verification without external evidence. The substrate is one tested way of supplying persistence and verification, not the only logically possible architecture.
    10. +
    11. Prior art and what is (and is not) claimed (byline, abstract, §1, §13). External memory, feedback-driven improvement and structured agent control have clear prior art: CoALA (Sumers et al., 2023, arXiv:2309.02427) organises language agents into modular memory, action and decision procedures, and Reflexion (Shinn et al., 2023, arXiv:2303.11366) improves agents with linguistic feedback and an episodic memory buffer and no weight updates. Both were already among this paper's references. The defensible claim is a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes; novelty is claimed only for a specific protocol, invariant, evaluation result or integration that survives an explicit prior-art comparison. The five faculties are a decomposition, not a proof of necessity, and the conclusion no longer says the architecture “supplies the faculties it structurally lacks”. The byline still called the three bodies of work “independently-developed” after the 2026-09-21 corrections had shown that they share an author; it now says so.
    12. +
    13. Reference grades are bibliographic (Appendix A). confirmed and traceable record that a source exists and is correctly attributed. They are not a judgement that the source supports the theory it is cited for. graded_reference_set.md now keeps bibliographic verification, claim support, study design, independent replication and transfer scope as separate grades.
    14. +
    +

    Appendix A — Graded reference set (new sources)

    -

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21]

    +

    The synthesis draws in a body of cognitive-architecture and process literature beyond the substrate paper's original 32 references. Each new source was independently verified this pass — modern arXiv sources by direct metadata fetch, classical works by primary-host search or established secondary knowledge — and graded: confirmed (record retrieved, attribution matches), traceable (the work clearly exists and is correctly attributed, but rests on established secondary knowledge rather than a single retrievable record), unverifiable (could not confirm). The tally: 9 confirmed, 6 traceable, 0 unverifiable (15 sources: 8 confirmed by the citations track plus the founding Agent-as-a-Judge paper added on its recommendation). Earlier versions said 8 confirmed. [corrected 2026-09-21] These grades are bibliographic: they confirm that a record exists and is correctly attributed. They do not grade whether a source supports the claim it is cited for, its study design, independent replication or transfer scope; research/formal-synthesis/graded_reference_set.md now keeps those as separate fields. [corrected 2026-09-26]

    SourceIDGradeNote
    Cognitive Architectures for Language Agents
    Theodore R. Sumers, Shunyu Yao, Karthik Narasi, 2023
    2309.02427confirmedRetrieved via arXiv metadata API; title/authors match claim exactly. Unifies memory, planning/reasoning, action, and learning modules into a single CoALA framework for language agents, giving the cognitive-substrate work's memory/im…
    diff --git a/research/python-prototypes/impact_oracle/README.md b/research/python-prototypes/impact_oracle/README.md index c3939095..3f66ba5a 100644 --- a/research/python-prototypes/impact_oracle/README.md +++ b/research/python-prototypes/impact_oracle/README.md @@ -56,8 +56,16 @@ in `oracle.py` as the module defaults: dependencies. Two terminal relations were added: `sibling` (one bounded forward hop to a bridge, then one bounded reverse hop from it, skipping bridges whose in-degree exceeds the cap) and `forward` (the changed symbol's own dependencies, ≤2 hops). - Held-out (never-tuned) repos at threshold 0.10: precision 0.320, recall 0.647, - **F1 0.428 vs the grep baseline's 0.371** — a reversal of the as-shipped 0.042 vs 0.437. + On the three held-out (never-tuned) repositories at the pre-registered canonical threshold + 0.02: precision 0.305, recall 0.653, **F1 0.416 against the grep baseline's 0.371** (ΔF1 about + +0.044). At threshold 0.10, which was selected on the tuning repositories, F1 is 0.428 + (precision 0.320, recall 0.647; ΔF1 +0.0565), and that is the only threshold with + per-repository counts: all three held-out repositories favour the repair there, but three of + three is a one-sided sign-test p of 0.125 and pytest supplies 71.3% of the held-out pairs, so + that it *beats* grep is not established. Never compare a 0.10 figure with a 0.02 one. (Corrected + 2026-09-26: this line quoted only the 0.10 result, as "F1 0.428 vs the grep baseline's 0.371 — a + reversal of the as-shipped 0.042 vs 0.437", which also set three held-out repositories against + all nine.) `ImpactOracle(wm, sibling_enabled=False, forward_enabled=False)` reproduces the as-shipped reverse-only traversal exactly; the untouched as-shipped package is archived as @@ -119,12 +127,16 @@ On this demo package the oracle reached recall 1.000 (it missed no affected modu these five mutations), with its best F1 of 0.79 at the optimal threshold (t=0.4). > **Refuted on real code.** That recall did not transfer. On 759 files in nine open-source -> Python repositories, with co-change ground truth, this version's recall was **0.022** and a +> Python repositories, with co-change ground truth, the as-shipped version's recall was **0.022** and a > grep baseline scored F1 0.437 against its 0.042: the traversal walks only reverse edges, and a -> construction defect breaks `src/`-layout packages. A repaired version ships in -> [`../../empirical-refutation/replication_package.tar.gz`](../../empirical-refutation/). Earlier -> versions of this README said the oracle "achieves perfect recall (never misses a truly affected -> module)". (Corrected 2026-09-21.) +> construction defect breaks `src/`-layout packages. This package **is** the repaired version (see +> "Repaired (v2)" above); the untouched as-shipped v1 is archived in +> [`../../empirical-refutation/replication_package.tar.gz`](../../empirical-refutation/), and +> `ImpactOracle(wm, sibling_enabled=False, forward_enabled=False)` reproduces its traversal. The +> results table above is the original five-mutation demonstration reported in the white paper (§8). Earlier versions of +> this README said the oracle "achieves perfect recall (never misses a truly affected module)". +> (Corrected 2026-09-21; the pointer to the repaired version corrected 2026-09-26 — it said "A +> repaired version ships in" the replication tarball.) ## File structure diff --git a/research/python-prototypes/router_gate/README.md b/research/python-prototypes/router_gate/README.md index 80cbd0df..0fda151f 100644 --- a/research/python-prototypes/router_gate/README.md +++ b/research/python-prototypes/router_gate/README.md @@ -173,3 +173,25 @@ python -m pytest ``` The included evaluation task set is a demonstration set, not a field benchmark. Calibrate thresholds on your own workload before enforcing automated model selection in high-stakes production paths. + +## Held-out result, and what "success" means + +*(Added 2026-09-26.)* The 30-task demonstration above was the set the thresholds were tuned on. On +80 pre-registered held-out tasks from real GitHub issues and pull requests +([`research/empirical-refutation/`](../../empirical-refutation/)), gate F1 was 0.37, and on the 64 +non-halted tasks the routed pipeline spent **$6.3582** against always-premium's **$5.2893** — +**20.21% more**, not saved. + +"Success" in that evaluation is **`judge_accepted`**: a model judge (the same model the pipeline +used as its mid-tier executor) accepted 6 of 64 routed outputs and 3 of 64 always-premium outputs. +That gives $1.060 against $1.763 per judge-accepted output, a ratio that is unstable at those counts. +No held-out task admitted execution-based verification, so **`tests_passed`**, **`human_accepted`** +and **`deployed_without_revert`** were never measured. Re-labelling 30 tasks with the same model and a +reworded prompt gave halt κ 0.5161 and tier κ 0.8919 — self-consistency, not agreement with a human. +The figures are recomputed from the replication archive by +[`research/recompute_corrections.py`](../../recompute_corrections.py) (§7, §9). + +When you evaluate this pipeline on your own workload, record those four outcomes separately, anchor +coding success on executable tests and blind human adjudication of disagreements, and keep gate +precision/recall, router solve rate and total pipeline cost as separate endpoints; count +clarification turns and tasks the gate rejected that were in fact valid. diff --git a/research/recompute_corrections.py b/research/recompute_corrections.py index 8d2c6ba8..4b89c248 100644 --- a/research/recompute_corrections.py +++ b/research/recompute_corrections.py @@ -8,11 +8,16 @@ python research/recompute_corrections.py rp/repro Sections 1-3 need no data (they are arithmetic on the synthesis paper's own worked -examples). Sections 4-9 read `results/*.json` and `data/*.parquet` from the package. +examples). Section 3b (added 2026-09-26) holds executable sanity checks on Theorem D that +need no data either; each is an `assert`, so a wrong statement fails the run with a +non-zero exit. Sections 4-9 read `results/*.json` and `data/*.parquet` from the package. Every random draw uses a fixed seed that is printed next to its result. -The corrections these numbers support were prompted by an external deep review of the -repository (2026-09-21); see the "Corrections" sections of each paper. + python research/recompute_corrections.py --theorem-checks # sections 1-3b only, no data + python research/recompute_corrections.py --help + +The corrections these numbers support were prompted by external deep reviews of the +repository (2026-09-21 and 2026-09-26); see the "Corrections" sections of each paper. """ import json @@ -117,6 +122,56 @@ def theorem_d(): print(f" T4: 150 lines x 80 bytes = {150 * 80} bytes > 8 KB cap (8192); 8192/150 = {8192 / 150:.1f} bytes/line") +def theorem_checks(): + """Executable sanity checks on Theorem D (2026-09-26 corrections). Each one is an assert.""" + header("3b. Theorem D sanity checks (asserted; 2026-09-26 corrections)") + tol = 1e-12 + + # (a) Separately maximal p and q need not be jointly attainable under one policy. The + # residual of a policy pi is (1 - p(pi)) * (1 - q(pi)); the target is its minimum over the + # joint feasible set F = {(p(pi), q(pi)) : admissible pi}, not the product of the maxima. + policies = {"A": (0.5, 0.9), "B": (0.9, 0.1)} + residual = {name: (1 - p) * (1 - q) for name, (p, q) in policies.items()} + p_max = max(p for p, _ in policies.values()) + q_max = max(q for _, q in policies.values()) + naive = (1 - p_max) * (1 - q_max) + attained = min(residual.values()) + for name, (p, q) in policies.items(): + print(f" policy {name}: (p, q) = ({p}, {q}) -> residual (1-p)(1-q) = {residual[name]:.4f}") + print(f" separate maxima p_max = {p_max}, q_max = {q_max} -> (1-p_max)(1-q_max) = {naive:.4f}") + print(f" lowest residual any feasible policy attains = {attained:.4f}") + assert abs(residual["A"] - 0.05) < tol, residual["A"] + assert abs(residual["B"] - 0.09) < tol, residual["B"] + assert abs(naive - 0.01) < tol, naive + assert naive < attained, "the product of separate maxima is not attained by any policy here" + print(" => (1-p_max)(1-q_max) is a LOWER bound on the attainable residual unless the maxima are") + print(" jointly attainable; optimise min over F = {(p(pi), q(pi))} of (1-p)(1-q) instead") + + # (b) Over n independent tasks with per-task residuals r_i <= eps, + # P(>=1 miss) = 1 - prod(1 - r_i) <= 1 - (1 - eps)^n, with EQUALITY only when every r_i = eps. + # Independence alone does not give equality; the union bound n * eps needs neither. + eps = 0.01 + unequal = [0.01, 0.005, 0.001] + n = len(unequal) + bound = 1 - (1 - eps) ** n + p_unequal = 1 - math.prod(1 - r for r in unequal) + p_equal = 1 - math.prod(1 - r for r in [eps] * n) + print(f" n={n}, eps={eps}: 1-(1-eps)^n = {bound:.6f}; union bound n*eps = {n * eps:.6f}") + print(f" residuals {unequal}: 1-prod(1-r_i) = {p_unequal:.6f} (strictly below the bound)") + print(f" residuals all equal to eps: 1-prod(1-r_i) = {p_equal:.6f} (equality)") + assert p_unequal < bound - tol, (p_unequal, bound) + assert abs(p_equal - bound) < tol, (p_equal, bound) + assert bound <= n * eps + tol, (bound, n * eps) + + # (c) Three lifecycle copies of one classifier are not three independent detectors: the + # independence product understates the nested residual (1-p)(1-c) 400-fold (section 2). + p, c, k = 0.7, 0.95, 3 + ratio = ((1 - p) * (1 - c)) / ((1 - p) * (1 - c) ** k) + print(f" {k} copies of one check (p={p}, c={c}): nested / independence-product residual = {ratio:.0f}x") + assert round(ratio) == 400, ratio + print(" all Theorem D sanity checks passed") + + # -------------------------------------------------------------------------------------- # Minimal Parquet reader (flat schema; PLAIN/dictionary; UNCOMPRESSED or SNAPPY) # -------------------------------------------------------------------------------------- @@ -344,6 +399,12 @@ def pooled(rows, idx): po, pg = pooled(oracle, full), pooled(grep, full) print(f" pooled oracle P/R/F1 = {po[0]:.4f} / {po[1]:.4f} / {po[2]:.4f}") print(f" pooled grep P/R/F1 = {pg[0]:.4f} / {pg[1]:.4f} / {pg[2]:.4f}") + # Repository-level view (added 2026-09-26): macro F1 weights each repository equally, so one + # large repository cannot carry the pooled figure. F1 is 0 where a method predicts nothing. + f1o = [prf(*oracle[i])[2] for i in full] + f1g = [prf(*grep[i])[2] for i in full] + print(f" macro F1 over {len(repos)} repositories: oracle {sum(f1o) / len(f1o):.4f}, grep {sum(f1g) / len(f1g):.4f}") + print(f" repositories where grep F1 > oracle F1: {sum(g > o for o, g in zip(f1o, f1g))} of {len(repos)}") rng = random.Random(SEED) draws = [] for _ in range(B): @@ -484,11 +545,18 @@ def w(i, j): def main(): + args = sys.argv[1:] + if args and args[0] in ("-h", "--help"): + print(__doc__) + return theorem_d() - if len(sys.argv) < 2: + theorem_checks() + if args and args[0] == "--theorem-checks": + return + if not args: print("\n(pass the extracted replication package's repro/ directory to recompute sections 4-9)") return - pkg = sys.argv[1] + pkg = args[0] ground_truth(pkg) cluster_bootstrap(pkg) repaired_vs_grep(pkg) diff --git a/scripts/build-pages.mjs b/scripts/build-pages.mjs index 4e774a87..c274a93d 100644 --- a/scripts/build-pages.mjs +++ b/scripts/build-pages.mjs @@ -237,7 +237,7 @@ h2{font-size:var(--fs-2);letter-spacing:-.01em;margin:0 0 var(--sp-4)} code{font:500 var(--fs-n1) var(--mono);background:var(--panel-2);border:1px solid var(--line);border-radius:var(--r-s);padding:0 var(--sp-2)} footer{padding:var(--sp-8) 0;color:var(--faint);font-size:var(--fs-n1)} @media(max-width:800px){.grid{grid-template-columns:1fr}} -

    ${esc(d.name)} · v${esc(d.version)} · Node ${esc(d.node)}

    Live status, straight from the repository.

    ${esc(d.description)}

    Install in 60 seconds Read the docs

    ${esc(d.license)} license${esc(d.deps)} runtime dependencies${esc(d.branch)} @ ${esc(d.commit)}${live}
    ${esc(d.impact)}
    blast-radius lookup

    Measured from this repo's benchmark report, not a marketing placeholder.

    reports/benchmarks.md

    ${esc(d.speed)}
    pre-action gate

    Assumptions, routing, reuse, context, impact, scope, and anchoring.

    reports/benchmarks.md

    ${esc(d.saved.match(/^[\d.]+\s*%?/)?.[0] ?? d.saved)}
    ${esc(d.saved.replace(/^[\d.]+\s*%?\s*/, "") || "routing signal")}

    Documented from the white-paper prototype and exposed by Forge cost reports.

    whitepaper prototype

    Quickstart

    npm install -g @codewithjuber/forgekit +

    ${esc(d.name)} · v${esc(d.version)} · Node ${esc(d.node)}

    Live status, straight from the repository.

    ${esc(d.description)}

    Install in 60 seconds Read the docs

    ${esc(d.license)} license${esc(d.deps)} runtime dependencies${esc(d.branch)} @ ${esc(d.commit)}${live}
    ${esc(d.impact)}
    blast-radius lookup

    Measured from this repo's benchmark report, not a marketing placeholder.

    reports/benchmarks.md

    ${esc(d.speed)}
    pre-action gate

    Assumptions, routing, reuse, context, impact, scope, and anchoring.

    reports/benchmarks.md

    ${esc(d.saved.match(/^[\d.]+\s*%?/)?.[0] ?? d.saved)}
    ${esc(d.saved.replace(/^[\d.]+\s*%?\s*/, "") || "routing signal")}

    Held-out result: on 80 pre-registered tasks the white-paper router spent this much more than always-premium; its 62.1% demo saving is refuted.

    research/empirical-refutation

    Quickstart

    npm install -g @codewithjuber/forgekit forge init forge doctor forge substrate "Change auth validation and update tests"

    Latest repo changes

      ${d.latest.map((x) => `
    • ${esc(x)}
    • `).join("")}

    Benchmark sections indexed: ${esc(d.benchMentions)} · benchmarks file updated ${esc(d.benchUpdated)}.

    Data Sources

    No mock data is used. This page is regenerated from repository files during CI (generated ${esc(d.generated)} from ${esc(d.commit)}). Enable BUILD_PAGES_LIVE=1 to refresh public GitHub counters with ETag/Last-Modified caching.

    • package.json
    • README.md
    • CHANGELOG.md
    • reports/benchmarks.md
    • ${api} (optional, no auth, only when BUILD_PAGES_LIVE=1)
    WCAG-minded semantic HTML, keyboard focus, responsive 320px–1920px+, and reduced-motion-safe. Same color and font tokens as the landing page — parity enforced in test/pages.test.js.
    `; diff --git a/scripts/claims-status.mjs b/scripts/claims-status.mjs new file mode 100644 index 00000000..24d0c8c1 --- /dev/null +++ b/scripts/claims-status.mjs @@ -0,0 +1,340 @@ +#!/usr/bin/env node +/** + * The claim/status registry, checked and rendered (node stdlib only). + * + * `docs/status/claims.json` records every load-bearing headline the project makes — what is + * claimed, which component and version it is about, the commit it was assessed against, its + * status, and the evidence behind it. This script keeps three things honest: + * + * 1. the registry itself: required fields, a closed set of statuses, unique ids, a commit + * that looks like one, and evidence paths that exist in the repository; + * 2. the status table in `docs/status/README.md`, which is generated from the registry + * between the CLAIMS:BEGIN / CLAIMS:END markers and never edited by hand; + * 3. the research copies under `docs/cognitive-substrate/`, which must stay byte-identical + * (sha256) to their canonical sources under `research/cognitive-substrate/`. + * + * Usage: + * node scripts/claims-status.mjs # validate, rewrite the table, check copies + * node scripts/claims-status.mjs --check # validate + fail on any drift (CI mode) + * node scripts/claims-status.mjs --sync-copies # re-copy canonical research files first + * node scripts/claims-status.mjs --root # operate on another checkout + * + * Exit codes: 0 ok · 1 invalid registry or drift · 2 usage error. + */ +import { createHash } from "node:crypto"; +import { copyFileSync, existsSync, readFileSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +/** The closed set of statuses a claim may carry, in display order. */ +export const STATUSES = ["implemented", "measured", "partial", "reported", "hypothesis", "refuted"]; + +/** Every claim must carry each of these fields. */ +export const REQUIRED_FIELDS = [ + "id", + "claim", + "component", + "version", + "source_commit", + "status", + "evidence", + "notes", +]; + +export const REGISTRY_PATH = "docs/status/claims.json"; +export const README_PATH = "docs/status/README.md"; +export const BEGIN = ""; +export const END = ""; + +/** + * Byte-identical copies kept under docs/ for readers there: [copy, canonical source]. + * `research/` is canonical; `--sync-copies` copies source → copy. + * @type {ReadonlyArray} + */ +export const COPY_PAIRS = [ + [ + "docs/cognitive-substrate/cognitive_substrate_whitepaper.html", + "research/cognitive-substrate/cognitive_substrate_whitepaper.html", + ], + [ + "docs/cognitive-substrate/cognitive_substrate_whitepaper.pdf", + "research/cognitive-substrate/cognitive_substrate_whitepaper.pdf", + ], + [ + "docs/cognitive-substrate/evidence_map.md", + "research/cognitive-substrate/evidence/evidence_map.md", + ], + [ + "docs/cognitive-substrate/ecosystem_map.md", + "research/cognitive-substrate/evidence/ecosystem_map.md", + ], +]; + +const ID_RE = /^[a-z0-9][a-z0-9-]*$/; +const COMMIT_RE = /^[0-9a-f]{7,40}$/; +const URL_RE = /^https?:\/\//; + +const DEFAULT_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); + +/** + * Validate a parsed registry. Pure apart from optional evidence-path existence checks, which + * run only when `root` is given. + * @param {any} registry the parsed claims.json + * @param {{root?: string|null}} [opts] + * @returns {string[]} human-readable problems; empty means valid + */ +export function validateRegistry(registry, { root = null } = {}) { + const errors = []; + if (!registry || typeof registry !== "object" || Array.isArray(registry)) { + return ["registry must be a JSON object with a `claims` array"]; + } + if (!Array.isArray(registry.claims) || registry.claims.length === 0) { + return ["registry.claims must be a non-empty array"]; + } + const seen = new Set(); + registry.claims.forEach((c, i) => { + const where = `claims[${i}]${c && typeof c.id === "string" ? ` (${c.id})` : ""}`; + if (!c || typeof c !== "object" || Array.isArray(c)) { + errors.push(`${where}: must be an object`); + return; + } + for (const f of REQUIRED_FIELDS) { + if (!(f in c)) errors.push(`${where}: missing required field "${f}"`); + } + for (const f of ["id", "claim", "component", "version", "source_commit", "status"]) { + if (f in c && (typeof c[f] !== "string" || !c[f].trim())) { + errors.push(`${where}: "${f}" must be a non-empty string`); + } + } + if ("notes" in c && typeof c.notes !== "string") { + errors.push(`${where}: "notes" must be a string`); + } + if (typeof c.id === "string" && c.id) { + if (!ID_RE.test(c.id)) errors.push(`${where}: id must match ${ID_RE}`); + if (seen.has(c.id)) errors.push(`${where}: duplicate id "${c.id}"`); + seen.add(c.id); + } + if (typeof c.status === "string" && c.status && !STATUSES.includes(c.status)) { + errors.push(`${where}: status "${c.status}" is not one of ${STATUSES.join(", ")}`); + } + if (typeof c.source_commit === "string" && c.source_commit) { + if (!COMMIT_RE.test(c.source_commit)) { + errors.push(`${where}: source_commit must be 7-40 lowercase hex characters`); + } + } + if ("evidence" in c) { + if (!Array.isArray(c.evidence) || c.evidence.length === 0) { + errors.push(`${where}: evidence must be a non-empty array of paths or URLs`); + } else { + for (const e of c.evidence) { + if (typeof e !== "string" || !e.trim()) { + errors.push(`${where}: every evidence entry must be a non-empty string`); + continue; + } + if (root && !URL_RE.test(e)) { + const rel = e.split("#")[0]; + if (!existsSync(path.join(root, rel))) { + errors.push(`${where}: evidence path not found: ${rel}`); + } + } + } + } + } + }); + return errors; +} + +/** Escape a value for a single Markdown table cell. */ +const cell = (s) => + String(s ?? "") + .replace(/\r?\n/g, " ") + .replace(/\|/g, "\\|") + .trim(); + +/** + * One evidence entry as a Markdown link: URLs keep their address; repository paths become + * links relative to docs/status/README.md. + * @param {string} e + */ +export function evidenceLink(e) { + if (URL_RE.test(e)) { + const label = e.replace(URL_RE, "").replace(/\/$/, ""); + return `[${cell(label)}](${e})`; + } + return `[\`${cell(e)}\`](../../${e})`; +} + +/** + * Render the generated block (without the markers): a status summary and one table row per + * claim, in registry order. + * @param {any} registry a valid registry + * @returns {string} + */ +export function renderTable(registry) { + const claims = registry.claims; + const counts = STATUSES.map((s) => [s, claims.filter((c) => c.status === s).length]); + const commits = [...new Set(claims.map((c) => c.source_commit))]; + const lines = [ + `${claims.length} claims — ${counts.map(([s, n]) => `${s} ${n}`).join(" · ")}.`, + `Assessed against commit${commits.length > 1 ? "s" : ""} ${commits.map((c) => `\`${c.slice(0, 12)}\``).join(", ")}${registry.as_of ? ` (as of ${registry.as_of})` : ""}.`, + "", + "| ID | Status | Claim | Component · version | Evidence | Notes |", + "| --- | --- | --- | --- | --- | --- |", + ]; + for (const c of claims) { + lines.push( + `| \`${cell(c.id)}\` | **${cell(c.status)}** | ${cell(c.claim)} | ${cell(c.component)} · ${cell(c.version)} | ${c.evidence.map(evidenceLink).join("
    ")} | ${cell(c.notes)} |`, + ); + } + return lines.join("\n"); +} + +/** + * Replace the generated block in the README text. + * @param {string} text current README + * @param {string} block rendered block (no markers) + * @returns {string} + */ +export function spliceReadme(text, block) { + const begin = text.indexOf(BEGIN); + const end = text.indexOf(END); + if (begin === -1 || end === -1 || end < begin) { + throw new Error(`${README_PATH} must contain the markers ${BEGIN} … ${END}`); + } + return `${text.slice(0, begin)}${BEGIN}\n${block}\n${text.slice(end)}`; +} + +/** @param {string} file */ +export const sha256File = (file) => createHash("sha256").update(readFileSync(file)).digest("hex"); + +/** + * Compare each docs copy with its canonical source. + * @param {string} root + * @param {ReadonlyArray} [pairs] + * @returns {{copy: string, source: string, ok: boolean, reason: string}[]} + */ +export function checkCopies(root, pairs = COPY_PAIRS) { + return pairs.map(([copy, source]) => { + const c = path.join(root, copy); + const s = path.join(root, source); + if (!existsSync(s)) return { copy, source, ok: false, reason: "canonical source missing" }; + if (!existsSync(c)) return { copy, source, ok: false, reason: "copy missing" }; + const same = sha256File(c) === sha256File(s); + return { copy, source, ok: same, reason: same ? "identical" : "sha256 differs" }; + }); +} + +/** + * Copy every drifted canonical source over its docs copy. + * @param {string} root + * @param {ReadonlyArray} [pairs] + * @returns {string[]} the copies rewritten + */ +export function syncCopies(root, pairs = COPY_PAIRS) { + const written = []; + for (const r of checkCopies(root, pairs)) { + if (r.ok || r.reason === "canonical source missing") continue; + copyFileSync(path.join(root, r.source), path.join(root, r.copy)); + written.push(r.copy); + } + return written; +} + +/** + * The CLI. Returns an exit code instead of exiting, so tests can drive it. + * @param {string[]} argv + * @param {{root?: string, pairs?: ReadonlyArray, + * log?: (s: string) => void, error?: (s: string) => void}} [io] + * @returns {number} + */ +export function run(argv, io = {}) { + const log = io.log ?? ((s) => process.stdout.write(`${s}\n`)); + const error = io.error ?? ((s) => process.stderr.write(`${s}\n`)); + let root = io.root ?? DEFAULT_ROOT; + const pairs = io.pairs ?? COPY_PAIRS; + const flags = new Set(); + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (a === "--root") { + const next = argv[++i]; + if (!next) { + error("--root needs a directory"); + return 2; + } + root = path.resolve(next); + } else if (a === "--check" || a === "--sync-copies") flags.add(a); + else if (a === "--help" || a === "-h") { + log( + "usage: node scripts/claims-status.mjs [--check] [--sync-copies] [--root ]\n" + + " (no flag) validate the registry, rewrite the table in docs/status/README.md, check copies\n" + + " --check validate and fail (exit 1) on a stale table or a drifted docs copy\n" + + " --sync-copies copy canonical research/ files over drifted docs/ copies first", + ); + return 0; + } else { + error(`unknown argument: ${a}`); + return 2; + } + } + const check = flags.has("--check"); + + let registry; + try { + registry = JSON.parse(readFileSync(path.join(root, REGISTRY_PATH), "utf8")); + } catch (e) { + error(`cannot read ${REGISTRY_PATH}: ${/** @type {Error} */ (e).message}`); + return 1; + } + const problems = validateRegistry(registry, { root }); + if (problems.length) { + for (const p of problems) error(`registry: ${p}`); + return 1; + } + + let failed = false; + const readmeFile = path.join(root, README_PATH); + let current; + try { + // Compare line content, not line endings: a Windows checkout (core.autocrlf) has CRLF. + current = readFileSync(readmeFile, "utf8").replace(/\r\n/g, "\n"); + } catch (e) { + error(`cannot read ${README_PATH}: ${/** @type {Error} */ (e).message}`); + return 1; + } + let expected; + try { + expected = spliceReadme(current, renderTable(registry)); + } catch (e) { + error(/** @type {Error} */ (e).message); + return 1; + } + if (expected !== current) { + if (check) { + error(`${README_PATH} is stale — run \`node scripts/claims-status.mjs\` to regenerate it`); + failed = true; + } else { + writeFileSync(readmeFile, expected); + log(`rewrote the generated table in ${README_PATH}`); + } + } + + if (flags.has("--sync-copies") && !check) { + for (const c of syncCopies(root, pairs)) log(`re-copied ${c} from its canonical source`); + } + for (const r of checkCopies(root, pairs)) { + if (!r.ok) { + error(`copy drift: ${r.copy} vs ${r.source} — ${r.reason} (run with --sync-copies)`); + failed = true; + } + } + + if (failed) return 1; + log( + `ok: ${registry.claims.length} claims valid, ${README_PATH} current, ${pairs.length} docs copies identical`, + ); + return 0; +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + process.exit(run(process.argv.slice(2))); +} diff --git a/src/cli.js b/src/cli.js index 555427f2..71352e6c 100755 --- a/src/cli.js +++ b/src/cli.js @@ -6,12 +6,15 @@ import { fileURLToPath } from "node:url"; // forge — zero-dependency dispatcher. Works identically whether installed via the // npm bin, the hardened install.sh symlink, or the Claude Code plugin. import { BRAND } from "./brand.js"; +// Domain command handlers live in their own modules (review A03); the presentation helpers +// every handler shares are defined once in ./cli/shared.js. +import memoryHandlers from "./cli/memory.js"; +import routingHandlers from "./cli/routing.js"; +import { bar, heading, paint, table } from "./cli/shared.js"; +import verificationHandlers from "./cli/verification.js"; // The command surface lives in commands.js as data — docs_check.js reconciles the // README/GUIDE tables against the same table this help is rendered from. import { COMMANDS, commandSummary, GROUPS } from "./commands.js"; -// Color is capability-gated (FORCE_COLOR > NO_COLOR > TERM=dumb > TTY) — piped -// output stays byte-plain, so nothing downstream ever parses an escape code. -import { bar, heading as fmtHeading, paint, table } from "./fmt.js"; import { printCommandHelp } from "./help.js"; import { suggest } from "./math.js"; @@ -39,14 +42,6 @@ function maybeFirstRunHint(cmd) { ); } -// Per-command title lines ("Forge — …") are branding chrome, not results. They -// print only when asked (`--verbose` or FORGE_VERBOSE=1); by default a command emits -// just its output. The `--help`/`--version` banner is unaffected. -const VERBOSE = process.argv.includes("--verbose") || process.env.FORGE_VERBOSE === "1"; -const heading = (text) => { - if (VERBOSE) console.log(fmtHeading(text)); -}; - function printHelp() { printVersion(); console.log(`\n${BRAND.tagline}\n`); @@ -65,7 +60,7 @@ function printHelp() { // Command dispatch table: name → async handler `(argv, cmd) => …`. Populated at module // load by the per-command handler declarations below. run() looks a command up here. /** @type {Record unknown>} */ -const HANDLERS = {}; +const HANDLERS = { ...memoryHandlers, ...verificationHandlers, ...routingHandlers }; async function run(argv) { const [cmd] = argv; @@ -350,6 +345,11 @@ HANDLERS.stack = async (argv) => { row("frameworks", s.frameworks); row("pkg mgrs", s.packageManagers); row("test", s.testCommands); + // Runners that are only installed (a devDependency) are inventory, never a required suite. + row( + "available", + (s.testInventory ?? []).filter((c) => !s.testCommands.includes(c)), + ); row("tools", s.tools); row("notes", s.notes); console.log(` ${"evidence:".padEnd(11)} ${s.evidence.join(", ")}`); @@ -718,578 +718,6 @@ HANDLERS.recall = async (argv) => { } return; }; -HANDLERS.ledger = async (argv) => { - const ls = await import("./ledger_store.js"); - const { epochDay, gitAuthor } = await import("./util.js"); - const root = process.cwd(); - // --personal targets the ledger beside the global recall store (~/.forge/recall/ - // ledger) — otherwise facts shadowed by `forge recall add` would be write-only, - // with no command able to inspect or verify them. - const personal = argv.includes("--personal"); - const args = argv.filter((a) => a !== "--json" && a !== "--personal"); - const dir = personal - ? (await import("node:path")).join((await import("./recall.js")).defaultStore(), "ledger") - : ls.repoLedger(root); - const sub = args[1] || "stats"; - const json = argv.includes("--json"); - const nowDay = epochDay(); - if (sub === "stats") { - const s = ls.stats(dir, nowDay); - if (json) return console.log(JSON.stringify(s, null, 2)); - heading(`${BRAND.brand} ledger — proof-carrying memory\n`); - console.log(` claims: ${s.total} (tombstoned ${s.tombstoned})`); - for (const [kind, n] of Object.entries(s.byKind)) console.log(` ${kind}: ${n}`); - if (s.pendingRetractions) - console.log( - paint( - ` ${s.pendingRetractions} claim(s) with an agent-proposed retraction — review, then \`forge ledger retract --reason …\``, - "warn", - ), - ); - console.log( - ` val: ${paint(`trusted ${s.val.trusted}`, "ok")} · ${paint(`uncertain ${s.val.uncertain}`, "warn")} · ${paint(`dormant ${s.val.dormant}`, "dim")}`, - ); - console.log( - paint("\n stored in .forge/ledger/ (git-committable, conflict-free merge)", "dim"), - ); - return; - } - if (sub === "verify") { - // --fix re-addresses claims still stored under their pre-CRLF-fold id. Reads accept - // that address either way, so this is not a repair — it is what stops one fact living - // at two addresses once a teammate on another platform mints its current form. - const migration = args.includes("--fix") ? ls.migrateAddresses(dir) : null; - const r = ls.verify(dir); - if (json) return console.log(JSON.stringify(migration ? { ...r, migration } : r, null, 2)); - if (migration) { - const { migrated, merged, failed } = migration; - console.log( - ` migrated ${migrated.length} claim(s) to their current address, merged ${merged.length} into an existing twin${failed.length ? `, ${failed.length} failed` : ""}`, - ); - } - console.log(` ${r.ok ? "OK" : "ISSUES"} — ${r.claims} claim(s), ${r.outcomes} outcome(s)`); - for (const i of r.issues) console.log(` - ${i}`); - if (!r.ok) process.exitCode = 1; - return; - } - if (sub === "show") { - const id = args[2]; - const hit = id && id.length >= 2 ? ls.getClaimByPrefix(dir, id, { attic: true }) : null; - if (!hit) { - console.error( - id ? ` no claim matching ${id}` : "usage: forge ledger show ", - ); - process.exitCode = 1; - return; - } - const { val } = await import("./ledger.js"); - const pending = ls.retractionProposals(ls.loadClaims(dir)).get(hit.id); - return console.log( - JSON.stringify( - { ...hit, val: val(hit, nowDay), ...(pending ? { pendingRetractions: pending } : {}) }, - null, - 2, - ), - ); - } - if (sub === "merge") { - const src = args[2]; - const { existsSync } = await import("node:fs"); - if (!src || !existsSync(src)) { - console.error( - src - ? ` no ledger at ${src}` - : "usage: forge ledger merge (a teammate's checkout, a backup, a worktree)", - ); - process.exitCode = 1; - return; - } - const r = ls.mergeDirs(dir, src); - if (json) return console.log(JSON.stringify(r, null, 2)); - console.log(` merged: ${r.claims} new claim(s), ${r.records} new record(s) — conflict-free`); - if (r.quarantined) - console.log( - ` quarantined: ${r.quarantined} invalid record(s) (forged hash or unresolvable ref — see quarantine/ in the ledger dir)`, - ); - return; - } - if (sub === "blame") { - const b = args[2] && args[2].length >= 2 ? ls.blame(dir, args[2], nowDay) : null; - if (!b) { - console.error( - args[2] ? ` no claim matching ${args[2]}` : "usage: forge ledger blame ", - ); - process.exitCode = 1; - return; - } - if (json) return console.log(JSON.stringify(b, null, 2)); - heading(`${BRAND.brand} ledger blame — ${b.kind} ${b.id.slice(0, 12)}\n`); - console.log( - ` val ${bar(b.val)} ${b.val.toFixed(2)} (trust-weighted ${b.valTrustWeighted.toFixed(2)})`, - ); - for (const p of b.minted) - console.log( - ` minted day ${p.t} by ${p.author || "(unknown)"}${p.agent ? ` · ${p.agent}` : ""}`, - ); - for (const e of b.evidence) - console.log( - ` ${e.result === "confirm" ? paint("confirm ", "ok") : paint("contradic", "err")} day ${e.t} ${e.oracle} → ${e.ref}${e.author ? ` by ${e.author}` : ""}`, - ); - for (const t of b.tombstones) - console.log( - paint(` retract day ${t.t} ${t.reason}${t.author ? ` by ${t.author}` : ""}`, "dim"), - ); - const trusts = Object.entries(b.trust); - if (trusts.length) { - console.log("\n author trust (earned from oracle outcomes on their claims):"); - for (const [a, u] of trusts) console.log(` ${u.toFixed(2)} ${a}`); - } - return; - } - // The two writes (08-dashboard-ux.md §2) — CLI twins of the dashboard's POSTs, so - // the dashboard stays a convenience, never a requirement. Both append-only. - if (sub === "ratify") { - const id = args[2]; - if (!id || id.length < 2) { - console.error("usage: forge ledger ratify "); - process.exitCode = 1; - return; - } - // Human-only promotion: the author is YOUR git identity, minted as a decision claim. - const r = ls.ratify(dir, id, { author: gitAuthor(), t: nowDay }); - if (!r.ok) { - console.error(` ${r.reason}`); - process.exitCode = 1; - return; - } - if (json) return console.log(JSON.stringify(r, null, 2)); - console.log( - ` ratified ${r.ratifies.slice(0, 12)} → decision ${r.decisionId.slice(0, 12)}${r.existed ? " (already ratified — same decision)" : ""}`, - ); - return; - } - if (sub === "retract") { - const id = args[2]; - const ri = args.indexOf("--reason"); - const reason = ri >= 0 ? (args[ri + 1] ?? "") : ""; - if (!id || id === "--reason" || !reason) { - console.error('usage: forge ledger retract --reason ""'); - process.exitCode = 1; - return; - } - // A tombstone is permanent, so it must name exactly one claim: the full 64-char id, - // never a prefix (a short prefix used to retract the first sorted match). - if (!ls.FULL_ID_RE.test(id)) { - console.error( - ` refused: retract needs the full 64-character claim id (got "${id}") — see \`forge ledger query\` or \`forge ledger show \``, - ); - process.exitCode = 1; - return; - } - const hit = ls.getClaimByPrefix(dir, id); - if (!hit) { - console.error(` no claim matching ${id}`); - process.exitCode = 1; - return; - } - const r = ls.tombstone(dir, hit.id, { - author: gitAuthor(), - reason, - t: nowDay, - }); - if (!r.ok) { - console.error(` ${r.reason}`); - process.exitCode = 1; - return; - } - if (json) return console.log(JSON.stringify({ ...r, id: hit.id }, null, 2)); - console.log( - ` retracted ${hit.id.slice(0, 12)} — ${reason}${r.deduped ? " (already retracted with this record)" : ""}`, - ); - return; - } - // `compact` — archive what this ledger's own history says will not be used again, and - // near-duplicates, printing every learned number (ledger_retention.js). Reversible. - if (sub === "compact") { - const dryRun = argv.includes("--dry-run"); - const r = ls.compactLedger(dir, nowDay, { dryRun }); - if (json) return console.log(JSON.stringify(r, null, 2)); - const rt = r.retention; - const d = r.duplicates; - const lines = [ - `Forge ledger — compact (every cut-off learned from this ledger)${dryRun ? " [dry run]" : ""}`, - "", - ` claims: ${r.claims} · claims with logged use: ${r.servedClaims}`, - rt.learned - ? ` retention: idle cut-off ${rt.cutoff} d = the longest idle stretch any claim came back from (${rt.comebacks} comebacks, typical gap ${rt.typicalGap} d; usage log spans ${rt.usageSpan} d)` - : ` retention: not learned — ${rt.reason}`, - d?.boundary != null - ? ` duplicates: boundary ${d.boundary.toFixed(2)} (two components beat one: BIC ${d.bic2?.toFixed(1)} < ${d.bic1?.toFixed(1)}) · ${d.groups.length} group(s)` - : ` duplicates: none — ${d?.compared ? `one component fits the ${d.compared} nearest-neighbour similarities better` : "fewer than two claims of one kind are still live to compare"}`, - "", - ` archive: ${r.archive.length}`, - ]; - for (const a of r.archive.slice(0, 20)) lines.push(` ${a.id.slice(0, 12)} ${a.reason}`); - if (r.archive.length > 20) lines.push(` … ${r.archive.length - 20} more (--json for all)`); - lines.push( - "", - dryRun - ? " dry run: nothing written" - : ` archived ${r.archived.length} claim(s) to .forge/ledger/attic/ — new evidence brings one back; show/blame still read it`, - ); - return console.log(lines.join("\n")); - } - if (sub === "query") { - const q = args.slice(2).join(" "); - if (!q) { - console.error('usage: forge ledger query ""'); - process.exitCode = 1; - return; - } - const { retrieve, claimText } = await import("./ledger.js"); - // The embeddings tier (ADR-0005) is assembled HERE, not in ledger.js — the pure - // core stays provider-free. No FORGE_EMBED (or a failing provider) → sim is - // null and retrieval is the stock MinHash path. - const { claimSim, simLabel } = await import("./embed.js"); - const claims = ls.loadClaims(dir); - const sim = claimSim(root, q, claims, claimText); - const ranked = retrieve(q, claims, { nowDay, budget: 8, sim }); - ls.recordUse( - dir, - ranked.map((r) => r.claim.id), - { via: "cli.query", t: nowDay }, - ); - if (json) - return console.log( - JSON.stringify( - { - sim: simLabel(sim), - results: ranked.map((r) => ({ - id: r.claim.id, - kind: r.claim.kind, - score: r.score, - })), - }, - null, - 2, - ), - ); - console.log(paint(` sim: ${simLabel(sim)}`, "dim")); - if (!ranked.length) return console.log(" no matching live claims"); - for (const r of ranked) - console.log( - ` ${bar(r.score, 8)} ${r.score.toFixed(3)} ${paint(r.claim.kind.padEnd(9), "accent")} ${paint(r.claim.id.slice(0, 8), "dim")} ${claimText(r.claim).slice(0, 90)}`, - ); - return; - } - // `at` / `diff` / `root` — the temporal surface. The store is append-only and every - // record carries its day, so a past day's beliefs are recomputed, never guessed. - const parseDay = (s) => { - if (/^\d{1,6}$/.test(s ?? "")) return Number(s); // bare epoch-day - const t = Date.parse(`${s}T00:00:00Z`); - if (Number.isNaN(t)) return null; - // Round-trip check: Date.parse silently rolls impossible dates over (2026-02-31 - // → March 3rd), which would answer a temporal query for a day nobody asked about. - if (new Date(t).toISOString().slice(0, 10) !== s) return null; - return Math.floor(t / 86_400_000); - }; - if (sub === "at") { - const day = parseDay(args[2]); - if (day === null) { - console.error(`usage: ${BRAND.cli} ledger at [--json]`); - process.exitCode = 1; - return; - } - const lg = await import("./ledger.js"); - const live = lg.liveClaims(lg.stateAt(ls.loadState(dir), day)); - const rows = live - .map((c) => ({ - id: c.id, - kind: c.kind, - val: Number(lg.val(c, day).toFixed(4)), - tombstoned: Boolean(c.tombstone), - text: lg.claimText(c).slice(0, 90), - })) - .sort((a, b) => b.val - a.val || (a.id < b.id ? -1 : 1)); - if (json) return console.log(JSON.stringify({ day, claims: rows.length, rows }, null, 2)); - heading(`${BRAND.brand} ledger — beliefs as of day ${day}\n`); - const byKind = {}; - for (const r of rows) byKind[r.kind] = (byKind[r.kind] ?? 0) + 1; - console.log( - ` claims: ${rows.length} (${rows.filter((r) => r.tombstoned).length} tombstoned)`, - ); - for (const [kind, n] of Object.entries(byKind)) console.log(` ${kind}: ${n}`); - for (const r of rows.slice(0, 10)) - console.log( - ` ${bar(r.val, 8)} ${r.val.toFixed(3)} ${paint(r.kind.padEnd(9), "accent")} ${paint(r.id.slice(0, 8), "dim")} ${r.text}`, - ); - return; - } - if (sub === "diff") { - const a = parseDay(args[2]); - const b = args[3] ? parseDay(args[3]) : nowDay; - if (a === null || b === null) { - console.error( - `usage: ${BRAND.cli} ledger diff [] [--json]`, - ); - process.exitCode = 1; - return; - } - if (a > b) { - // beliefDiff's contract is dayA ≤ dayB; a reversed window would print silently - // inverted appeared/retired classes, so refuse loudly instead. - console.error(` (day ${a}) is after (day ${b}) — swap the arguments`); - process.exitCode = 1; - return; - } - const lg = await import("./ledger.js"); - const d = lg.beliefDiff(ls.loadState(dir), a, b); - if (json) return console.log(JSON.stringify({ since: a, until: b, ...d }, null, 2)); - heading(`${BRAND.brand} ledger — what changed, day ${a} → ${b}\n`); - console.log( - ` appeared ${d.appeared.length} · retired ${d.retired.length} · ${paint(`strengthened ${d.strengthened.length}`, "ok")} · ${paint(`weakened ${d.weakened.length}`, "warn")}`, - ); - const row = (label, r) => - console.log( - ` ${label} ${paint(r.kind.padEnd(9), "accent")} ${paint(r.id.slice(0, 8), "dim")} ${r.from === null ? "· " : r.from.toFixed(2)} → ${r.to === null ? "·" : r.to.toFixed(2)} ${r.text.slice(0, 70)}`, - ); - for (const r of d.appeared.slice(0, 5)) row(paint("new ", "ok"), r); - for (const r of d.retired.slice(0, 5)) row(paint("gone", "err"), r); - for (const r of d.strengthened.slice(0, 5)) row(paint("up ", "ok"), r); - for (const r of d.weakened.slice(0, 5)) row(paint("down", "warn"), r); - return; - } - if (sub === "root") { - const lg = await import("./ledger.js"); - const r = lg.stateRoot(ls.loadState(dir)); - if (json) return console.log(JSON.stringify(r, null, 2)); - console.log(r.root); // bare hex on stdout — scriptable ("are we in sync?" is one diff) - return; - } - if (sub === "sync") { - const { ledgerSync, defaultRef } = await import("./ledger_sync.js"); - const di = args.indexOf("--dir"); - const re = args.indexOf("--remote"); - const rf = args.indexOf("--ref"); - const r = ledgerSync({ - dir, - root, - personal, - dirTarget: di >= 0 ? args[di + 1] : undefined, - remote: re >= 0 ? args[re + 1] : undefined, - ref: rf >= 0 ? args[rf + 1] : undefined, - }); - if (json) return console.log(JSON.stringify(r, null, 2)); - heading(`${BRAND.brand} ledger sync\n`); - if (!r.ok) { - console.error(` ${paint(r.reason ?? "sync failed", "err")}`); - process.exitCode = 1; - return; - } - if (r.mode === "dir") { - console.log( - table([ - [paint("target", "dim"), r.dir], - [paint("pulled", "dim"), `${r.pulled.claims} claim(s), ${r.pulled.records} record(s)`], - [ - paint("pushed", "dim"), - r.upToDate - ? paint("up to date — state roots match, nothing to merge", "dim") - : `${r.pushed.claims} claim(s), ${r.pushed.records} record(s)`, - ], - ]), - ); - } else { - console.log( - table([ - [paint("ref", "dim"), `${r.remote} ${r.ref}`], - [paint("pulled", "dim"), `${r.pulled.claims} claim(s), ${r.pulled.records} record(s)`], - [ - paint("pushed", "dim"), - r.upToDate - ? paint("up to date — nothing to push", "dim") - : `yes (retries ${r.retries})`, - ], - ]), - ); - } - for (const n of r.notes ?? []) console.log(paint(` note: ${n}`, "warn")); - if (r.mode === "ref" && r.ref === defaultRef(personal)) - console.log(paint("\n synced through a git ref — CRDT, converges in any order", "dim")); - return; - } - if (sub === "import") { - const b = await import("./ledger_bridge.js"); - let r; - if (personal) { - // Personal import: facts from the global recall store into the personal ledger. - const { defaultStore } = await import("./recall.js"); - r = { - lessons: 0, - outcomes: 0, - ...b.importFacts(defaultStore(), dir, nowDay), - }; - } else { - const { brainStore } = await import("./brain.js"); - r = b.importLegacy(root, { - recallStore: brainStore(root), - recallLedger: dir, - nowDay, - }); - } - if (json) return console.log(JSON.stringify(r, null, 2)); - console.log(` imported: ${r.lessons} lesson(s), ${r.facts} fact(s), ${r.outcomes} outcome(s)`); - for (const x of r.refused) console.log(` refused: ${x}`); - return; - } - console.error( - `ledger: unknown subcommand "${sub}" (stats | verify | show | blame | query | at | diff [] | root | ratify | retract --reason "" | merge | sync [--dir |--remote |--ref ] | import) [--personal] [--json]`, - ); - process.exitCode = 1; - return; -}; -HANDLERS.reuse = async (argv) => { - const ru = await import("./reuse.js"); - const { load: loadAtlas } = await import("./atlas.js"); - const { epochDay } = await import("./util.js"); - const root = process.cwd(); - const nowDay = epochDay(); - const json = argv.includes("--json"); - const flagVal = (name) => { - const i = argv.indexOf(name); - return i >= 0 ? argv[i + 1] : undefined; - }; - const args = argv.filter( - (a, i) => !a.startsWith("--") && argv[i - 1] !== "--file" && argv[i - 1] !== "--ref", - ); - const sub = args[1] || "stats"; - if (sub === "query") { - const spec = args.slice(2).join(" "); - if (!spec) { - console.error('usage: forge reuse query "" [--json]'); - process.exitCode = 1; - return; - } - const r = ru.reuseQuery(root, spec, { atlas: loadAtlas(root), nowDay }); - if (json) - return console.log( - JSON.stringify( - { - tier: r.tier, - artifact: r.artifact?.id, - jaccard: r.jaccard, - similarity: r.similarity, - sim: r.sim, - reasons: r.reasons, - }, - null, - 2, - ), - ); - console.log(` sim: ${r.sim}`); - if (r.tier === "miss") { - console.log(" miss — nothing verified matches; generate, then `forge reuse mint` it"); - } else { - const a = r.artifact; - console.log( - ` ${r.tier.toUpperCase()} hit (similarity ${(r.similarity ?? r.jaccard ?? 1).toFixed(2)}) — ${a.body.form}${a.body.code?.path ? ` at ${a.body.code.path}` : ""}`, - ); - console.log( - ` claim ${a.id.slice(0, 12)} — \`forge ledger blame ${a.id.slice(0, 8)}\` for its proof`, - ); - if (r.tier === "adapt") - console.log(" adapt tier: inject as a verified starting point, generate only the delta"); - } - for (const why of r.reasons) console.log(` note: ${why}`); - return; - } - if (sub === "mint") { - const spec = args.slice(2).join(" "); - const file = flagVal("--file"); - const ref = flagVal("--ref"); - if (!spec || !file) { - console.error( - 'usage: forge reuse mint "" --file [--ref ] [--json]', - ); - process.exitCode = 1; - return; - } - const { repoLedger } = await import("./ledger_store.js"); - const desc = ru.describeFile(root, file); - const r = ru.mintArtifact( - repoLedger(root), - { spec, form: "module", ...desc }, - ref - ? { - evidence: { oracle: "test.run", result: "confirm", ref }, - t: nowDay, - } - : { t: nowDay }, - ); - if (json) return console.log(JSON.stringify(r, null, 2)); - if (!r.ok) { - console.error(` ${r.reason}`); - process.exitCode = 1; - return; - } - console.log( - ` minted: ${r.id.slice(0, 12)} (${desc.iface.length} export(s), ${desc.deps.length} dep(s))`, - ); - console.log( - r.serves - ? " serving: yes — verification evidence attached" - : " serving: NOT YET — no evidence; attach a verified test/commit ref (--ref) or it stays at the 0.5 prior", - ); - return; - } - if (sub === "stats") { - const { summarize } = await import("./metrics.js"); - const s = summarize(root).cache ?? { - events: 0, - byOutcome: {}, - savedEstimate: 0, - }; - if (json) return console.log(JSON.stringify(s, null, 2)); - heading(`${BRAND.brand} reuse — proof-carrying code cache\n`); - console.log(` lookups: ${s.events}`); - for (const [o, n] of Object.entries(s.byOutcome)) console.log(` ${o}: ${n}`); - console.log(` est. tokens saved: ${s.savedEstimate}`); - return; - } - console.error( - `reuse: unknown subcommand "${sub}" (query | mint --file | stats)`, - ); - process.exitCode = 1; - return; -}; -HANDLERS.context = async (argv) => { - const { assemble, renderContext } = await import("./context.js"); - const { load: loadAtlas } = await import("./atlas.js"); - const { epochDay } = await import("./util.js"); - const json = argv.includes("--json"); - const bi = argv.indexOf("--budget"); - const budget = bi >= 0 ? Number(argv[bi + 1]) || undefined : undefined; - const task = argv - .filter((a, i) => i > 0 && !a.startsWith("--") && argv[i - 1] !== "--budget") - .join(" "); - if (!task) { - console.error('usage: forge context "" [--budget ] [--json]'); - process.exitCode = 1; - return; - } - const r = assemble(process.cwd(), task, { - atlas: loadAtlas(process.cwd()), - nowDay: epochDay(), - ...(budget ? { budget } : {}), - }); - if (json) { - const { block, ...rest } = r; - return console.log(JSON.stringify(rest, null, 2)); - } - console.log(renderContext(r)); - if (!r.ok) process.exitCode = 1; - return; -}; HANDLERS.atlas = async (argv) => { const a = await import("./atlas.js"); const sub = argv[1] || "build"; @@ -1415,127 +843,13 @@ HANDLERS.scan = async (argv) => { if (r.critical) process.exitCode = 1; return; }; -HANDLERS.verify = async (argv) => { - const json = argv.includes("--json"); - if (argv.includes("--deep")) { - const { verifyDeep, LENSES } = await import("./consensus.js"); - // `--llm` opts the reviewer panel in for this run; otherwise FORGE_LLM decides. - const r = verifyDeep({ - targetRoot: process.cwd(), - llm: argv.includes("--llm") ? true : undefined, - }); - if (json) { - console.log(JSON.stringify(r, null, 2)); - if (!r.ok) process.exitCode = 1; - return; - } - heading(`${BRAND.brand} verify --deep — multi-lens consensus\n`); - console.log( - table( - r.lenses.map((l) => { - const meta = LENSES[l.lens]; - const state = - l.ran === false - ? paint("— skipped", "dim") - : l.s > 0 - ? paint("● finding", meta.solo ? "err" : "warn") - : paint("✓ clean", "ok"); - return [l.lens, meta.family, `w=${meta.weight}`, state]; - }), - ), - ); - if (r.findings.length) { - console.log(); - for (const f of r.findings) console.log(` ! ${f}`); - } - console.log( - `\n defectRiskScore: ${bar(r.p)} ${r.p.toFixed(2)} (heuristic, not a calibrated probability)${ - r.families.length ? ` (families: ${r.families.join(", ")})` : "" - }`, - ); - console.log( - ` remainingUncheckedWeight: ${r.residual.toFixed(3)} — Theorem-D silent-miss bound (heuristic)`, - ); - // The core tests state, spelled out — deep ok REQUIRES a passing core, so the - // reader must see whether a verifier actually ran (RA-01). - const t = r.tests ?? /** @type {import("./verify.js").VerifyTests} */ ({ ran: false }); - console.log( - ` tests: ${t.status ?? "unknown"}${t.runner ? ` (${t.runner})` : ""}${ - !t.runner && t.detected?.length ? ` (detected: ${t.detected.join(", ")})` : "" - }`, - ); - if (t.executed?.length) - console.log( - ` suites ran: ${t.executed.map((s) => `${s.label}=${s.status}`).join(", ")}`, - ); - if (t.notExecuted?.length) - console.log(` suites skipped: ${t.notExecuted.join(", ")} (no built-in executor)`); - const detail = t.output || (t.detected ?? []).join(", "); - const verdictLine = - r.status === "PASS" - ? paint("PASS", "ok") - : r.status === "NOT_CONFIGURED" - ? paint("NOT VERIFIED — no test runner configured (NOT_CONFIGURED)", "warn") - : r.status === "INCOMPLETE" - ? paint(`NOT VERIFIED — tests incomplete${detail ? ` (${detail})` : ""}`, "warn") - : paint("BLOCKED — cross-family consensus says defect", "err"); - console.log(`\n ${verdictLine}`); - if (!r.ok) process.exitCode = 1; - return; - } - const { verify } = await import("./verify.js"); - const r = verify({ targetRoot: process.cwd() }); - if (json) { - console.log(JSON.stringify(r, null, 2)); - if (!r.ok) process.exitCode = 1; - return; - } - heading(`${BRAND.brand} verify\n`); - console.log(` changed files: ${r.changedFiles.length}`); - // Honest four-state tests line (RA-09): "nothing ran" is never dressed up as a pass, - // and a real run names the runner that actually executed. - const t = r.tests; - const testsLine = - t.status === "PASS" - ? `✓ pass (${t.runner ?? "project suite"})` - : t.status === "FAIL" - ? `✗ FAIL (${t.runner ?? "project suite"})` - : t.status === "INCOMPLETE" - ? `— INCOMPLETE: ${t.output || (t.detected ?? []).join(", ") || "test run did not complete"}` - : "— NOT CONFIGURED (no test runner detected)"; - console.log(` tests: ${testsLine}`); - // Per-suite honesty (HI-01): a polyglot repo runs every executable suite; name which - // ones ran with what verdict, and which were detected but never executed. - if (t.executed?.length) - console.log( - ` suites ran: ${t.executed.map((s) => `${s.label}=${s.status}`).join(", ")}`, - ); - if (t.notExecuted?.length) - console.log(` suites skipped: ${t.notExecuted.join(", ")} (no built-in executor)`); - console.log(` symbols checked: ${r.provenance.symbolsChecked}`); - if (r.unknown.length) - console.log( - ` ! not in codebase (possible hallucination): ${r.unknown.slice(0, 12).join(", ")}`, - ); - console.log(` provenance: .forge/provenance.json`); - // BLOCKED is reserved for a runner that actually FAILED; anything that never ran - // to completion is NOT VERIFIED (still exit 1 — unverified is not a pass). - const verdict = r.ok - ? "PASS" - : t.status === "FAIL" - ? "BLOCKED — tests failing" - : `NOT VERIFIED — ${t.status}`; - console.log(`\n ${verdict}`); - if (!r.ok) process.exitCode = 1; - return; -}; -HANDLERS.remember = async (argv) => { - const b = await import("./brain.js"); - const name = argv[1]; - const body = argv.slice(2).join(" "); - if (!name || !body) { - console.error('usage: forge remember "" ""'); - process.exitCode = 1; +HANDLERS.remember = async (argv) => { + const b = await import("./brain.js"); + const name = argv[1]; + const body = argv.slice(2).join(" "); + if (!name || !body) { + console.error('usage: forge remember "" ""'); + process.exitCode = 1; return; } const res = b.remember(b.brainStore(process.cwd()), name, body); @@ -1574,117 +888,6 @@ HANDLERS.brain = async () => { ); return; }; -HANDLERS.cost = async (argv) => { - // `--stages` is the P8 measured report (per-stage factors from .forge/metrics.jsonl); - // the default path stays the ccusage per-day spend view, untouched. - if (argv.includes("--stages")) { - const { renderCostReport, report } = await import("./cost_report.js"); - const r = report(process.cwd()); - console.log(argv.includes("--json") ? JSON.stringify(r, null, 2) : renderCostReport(r)); - return; - } - const { execFileSync } = await import("node:child_process"); - const run = (bin, args) => execFileSync(bin, args, { encoding: "utf8", stdio: "pipe" }); - heading(`${BRAND.brand} cost — real per-day spend (ccusage)\n`); - try { - let out; - try { - out = run("ccusage", ["daily"]); - } catch { - // Pinned (verified 2026-07-05) — never @latest for code we execute; re-verify via dev-radar. - out = run("npx", ["-y", "ccusage@20.0.14", "daily"]); - } - console.log(out.trim()); - } catch { - const { estimateSpendFromLogs } = await import("./cost_report.js"); - const est = estimateSpendFromLogs({ root: process.cwd() }); - if (est && est.totalCost > 0) { - console.log( - ` $${est.totalCost.toFixed(2)} estimated from Claude session logs (${est.sessions} session(s))`, - ); - if (est.byModel.length) { - for (const m of est.byModel) - console.log( - ` ${m.model.padEnd(30)} ${m.priced ? `$${m.cost.toFixed(4)}` : "unpriced"} (${m.inTokens} in / ${m.outTokens} out)${m.priceSource ? ` · price: ${m.priceSource}` : ""}`, - ); - } - if (est.unpriced?.length) - console.log( - paint( - ` not in the total — no catalog or snapshot price for: ${est.unpriced.join(", ")}`, - "dim", - ), - ); - console.log(paint("\n install ccusage for precise tracking: npm i -g ccusage", "dim")); - } else { - console.log( - " ccusage not found. Install for real spend (reads local JSONL, nothing leaves your machine):\n npm i -g ccusage # then: forge cost", - ); - } - } - console.log( - paint( - `\n ceiling: FORGE_COST_CEILING (default $10) — the cost-budget guard warns when a day exceeds it.`, - "dim", - ), - ); - return; -}; -HANDLERS.models = async (argv) => { - // What each tier resolves to RIGHT NOW: family → newest model in the active provider's live - // catalog (else the shipped snapshot), priced from OpenRouter's catalog (else the snapshot). - const { describeResolution, PRICING_VERIFIED, resolveTiers } = await import("./model_tiers.js"); - const { activeProvider, envModelOverride } = await import("./providers.js"); - const root = process.cwd(); - const provider = activeProvider(root); - const tiers = resolveTiers({ root, provider }); - const override = envModelOverride(); - if (argv.includes("--json")) - return console.log( - JSON.stringify( - { provider: provider.name, override, pricingVerified: PRICING_VERIFIED, tiers }, - null, - 2, - ), - ); - heading(`${BRAND.brand} models — each tier's family, resolved to a concrete model\n`); - console.log(` provider ${provider.name} (${provider.label || provider.name})`); - if (override) - console.log( - ` override ${override} — ANTHROPIC_MODEL/FORGE_MODEL pins every call; the tiers below apply without it`, - ); - const priceText = (p) => (p ? `$${p.inCost}/$${p.outCost}` : "—"); - const priceFrom = (p) => - !p - ? "unpriced" - : p.source === "catalog" - ? "catalog" - : `snapshot${p.basis === "family" ? " (tier)" : ""}`; - const width = Math.max(28, ...tiers.map((t) => (t.model?.id ?? "").length + 2)); - console.log( - `\n ${"tier".padEnd(8)} ${"family".padEnd(7)} ${"model".padEnd(width)} ${"created".padEnd(11)} ${"$/M tok".padEnd(10)} ${"id from".padEnd(9)} price from`, - ); - for (const t of tiers) { - console.log( - ` ${t.class.padEnd(8)} ${t.family.padEnd(7)} ${(t.model?.id ?? "—").padEnd(width)} ${(t.model?.createdAt?.slice(0, 10) ?? "—").padEnd(11)} ${priceText(t.price).padEnd(10)} ${(t.model?.source ?? "—").padEnd(9)} ${priceFrom(t.price)}`, - ); - } - console.log(""); - for (const t of tiers) console.log(` ${t.family.padEnd(7)} ${describeResolution(t.model)}`); - const live = tiers.find((t) => t.price?.source === "catalog")?.price; - console.log( - live - ? `\n prices: live from ${live.catalog}${live.cache === "stale" ? " (last cached copy — catalog unreachable)" : ""}` - : `\n prices: shipped snapshot, verified ${PRICING_VERIFIED} (OpenRouter's catalog unavailable or unlisted)`, - ); - console.log( - paint( - " cache: .forge/cache/ — reused while the response's own Cache-Control/Expires says fresh, else revalidated (ETag); FORGE_NO_CATALOG_FETCH=1 stays offline", - "dim", - ), - ); - return; -}; HANDLERS.spec = async (argv) => { const s = await import("./speclock.js"); const sub = argv[1] || "check"; @@ -2081,229 +1284,6 @@ HANDLERS.config = async (argv) => { process.exitCode = 1; return; }; -HANDLERS.route = async (argv) => { - const r = await import("./route.js"); - if (argv[1] === "gateway") { - const result = r.emitGatewayConfig(process.cwd()); - if (typeof result === "object" && !result.ok) { - console.log(` ${result.reason}`); - return; - } - console.log(` wrote ${result} — LiteLLM tiers: forge-simple / forge-medium / forge-complex.`); - console.log(" next: pin+install litellm, run it, point ANTHROPIC_BASE_URL at it, then"); - console.log( - " REQUEST the tier `forge route` recommends (a plain claude-* request passes through).", - ); - return; - } - if (argv[1] === "calibrate") { - // Advisory → gated promotion (ROADMAP): measure whether an affine calibration of the - // routing rubric beats the raw rubric on a held-out split of the HAND-LABELLED fixture - // (there is no outcome data). Advisory — routing keeps the rubric unless the gate - // promotes AND a caller adopts the calibration, which nothing in src/ does. - const res = r.calibrateRouting(); - if (argv.includes("--json")) return console.log(JSON.stringify(res, null, 2)); - heading(`${BRAND.brand} route calibrate — rubric calibration check (measured gate)\n`); - console.log(` samples: ${res.n} hand-labelled task phrase(s) — no routing outcomes exist`); - if (res.baselineMetric !== undefined) - console.log( - ` held-out MAE: rubric ${res.baselineMetric} · calibrated ${res.candidateMetric}`, - ); - console.log( - res.mode === "candidate" - ? ` → PROMOTE calibration — ${res.reason} (a=${res.model.a.toFixed(3)}, b=${res.model.b.toFixed(3)})` - : ` → keep the rubric — ${res.reason}`, - ); - console.log( - "\n advisory — routing stays on the rubric; nothing adopts a promoted calibration yet,", - ); - console.log(" and calibrating on real routing outcomes needs data forge does not record"); - return; - } - if (["universal", "outcome", "fit", "models"].includes(argv[1]) || argv.includes("--universal")) { - return routeUniversalCli(argv); - } - const json = argv.includes("--json"); - const apply = argv.includes("--apply"); - const providerIdx = argv.indexOf("--provider"); - const providerName = providerIdx >= 0 ? argv[providerIdx + 1] : undefined; - const FLAGS = new Set(["--json", "--apply"]); - const task = argv - .slice(1) - .filter((a, i) => !FLAGS.has(a) && a !== "--provider" && argv[i] !== "--provider") - .join(" "); - if (!task) { - console.error( - 'usage: forge route "" [--apply] [--provider ] [--json] | forge route gateway', - ); - process.exitCode = 1; - return; - } - if (providerName) { - const { setProvider } = await import("./providers.js"); - const sr = setProvider(process.cwd(), providerName); - if (!sr.ok) { - console.error(` ${sr.reason}`); - process.exitCode = 1; - return; - } - } - const rec = r.routeTask(process.cwd(), task); - r.meterRoute(process.cwd(), task, rec); - // The recommendation is a tier (a model family); its concrete id and price are resolved here, - // in the command — routeTask stays network-free because the hooks run it. BOTH output modes - // resolve, so a script reading --json never sees a different model than the text prints. - const { describeResolution, resolveTierModel, resolveTierPrice } = await import( - "./model_tiers.js" - ); - const { activeProvider } = await import("./providers.js"); - const opts = { root: process.cwd(), provider: activeProvider(process.cwd()) }; - const resolved = resolveTierModel(rec.key, opts); - const price = resolveTierPrice(rec.key, { ...opts, resolved }); - if (json) { - // `model` stays the snapshot row (its shape is public); `resolved` is what would be called. - console.log(JSON.stringify({ ...rec, resolved, price }, null, 2)); - } else { - heading(`${BRAND.brand} route — cheapest capable model\n`); - const name = - resolved?.source === "catalog" && resolved.displayName - ? resolved.displayName - : rec.model.name; - console.log( - ` → ${paint(name, "accent")} (${rec.tier}, ${price ? `${price.inCost}/${price.outCost} per M tok, ${price.source === "catalog" ? "live price" : "current effective"}` : "price unknown"})`, - ); - if (resolved) console.log(` model: ${resolved.id} — ${describeResolution(resolved)}`); - console.log(` ${rec.model.use}`); - console.log( - ` complexity ${bar(rec.score, 8)} ${rec.score.toFixed(2)}${rec.reasons.length ? ` · driven by: ${rec.reasons.join(", ")}` : ""}`, - ); - console.log( - ` signals: ${rec.signals.files} file(s), fan-out ${rec.signals.fanout}, churn ${rec.signals.churn}, past-mistakes ${rec.signals.pastMistakes}, ambiguity ${rec.signals.ambiguity.toFixed(2)}`, - ); - } - if (apply) { - const { applyRoute } = await import("./providers.js"); - const ar = applyRoute(rec.key); - if (ar.ok) { - if (!json) console.log(`\n applied: model set to ${ar.model} (${ar.modelId}) in ${ar.path}`); - } else { - if (!json) console.error(`\n apply failed: ${ar.reason}`); - process.exitCode = 1; - } - } else if (!json) { - console.log( - `\n advisory · apply: \`${BRAND.cli} route "" --apply\` · gateway: \`${BRAND.cli} route gateway\``, - ); - } - return; -}; -// Universal router (src/router): any provider's models, chosen by expected cost for the success -// probability asked for. Models come from data/models.json and .forge/models.json. -async function routeUniversalCli(argv) { - const U = await import("./router/index.js"); - const { loadRegistry } = await import("./router/registry.js"); - const json = argv.includes("--json"); - const val = (flag) => (argv.includes(flag) ? argv[argv.indexOf(flag) + 1] : undefined); - const VALUED = new Set(["--objective", "--provider", "--model", "--cost", "--depth"]); - const words = argv - .slice(1) - .filter((a, i, arr) => !a.startsWith("--") && !VALUED.has(arr[i - 1] ?? "")); - const sub = ["outcome", "fit", "models", "universal"].includes(words[0]) - ? words.shift() - : "universal"; - const root = process.cwd(); - if (sub === "models") { - const reg = loadRegistry(root); - const fit = U.loadRouterModel(root); - const rows = reg.models.map((m) => ({ - id: m.id, - org: m.org ?? null, - status: fit?.models.includes(m.id) ? "fitted" : "cold", - providers: Object.keys(m.providers ?? {}), - price: m.price_in != null ? `${m.price_in}/${m.price_out}` : null, - })); - if (json) - return console.log( - JSON.stringify({ sources: reg.sources, fit: fit?.origin ?? null, models: rows }, null, 2), - ); - heading(`${BRAND.brand} route models — registry (${reg.sources.join(" + ")})\n`); - for (const r of rows) - console.log( - ` ${r.id.padEnd(22)} ${String(r.org ?? "").padEnd(16)} ${r.status.padEnd(7)} ${r.price ? `$${r.price}/Mtok`.padEnd(14) : "".padEnd(14)} ${r.providers.join(", ") || "(no provider id: advice only)"}`, - ); - console.log(`\n fit in use: ${fit?.origin ?? "none"}`); - return; - } - if (sub === "fit") { - const model = U.fitRouter(root); - if (json) return console.log(JSON.stringify(model.provenance, null, 2)); - console.log( - ` refit on ${model.provenance.local.outcomes} recorded outcome(s) over ${model.provenance.local.tasks} task(s); wrote .forge/router_model.json`, - ); - return; - } - const task = words.join(" "); - if (!task) { - console.error( - 'usage: forge route universal "" [--objective match-best-single|target:

    |value:<$>|budget:<$>] [--provider |any] [--depth ] [--json]\n' + - ' forge route outcome "" --model --pass|--fail [--cost ]\n' + - " forge route fit | forge route models", - ); - process.exitCode = 1; - return; - } - if (sub === "outcome") { - const passed = argv.includes("--pass") ? true : argv.includes("--fail") ? false : undefined; - const cost = val("--cost") !== undefined ? Number(val("--cost")) : null; - try { - const row = U.recordOutcome(root, { task, model: val("--model"), passed, cost }); - if (json) return console.log(JSON.stringify(row, null, 2)); - console.log( - ` recorded ${row.model} ${row.passed ? "pass" : "fail"} for task ${row.task} (.forge/route_outcomes.jsonl)`, - ); - } catch (e) { - console.error(` ${e.message}`); - process.exitCode = 1; - } - return; - } - let rec; - try { - rec = U.routeUniversal(root, task, { - objective: val("--objective"), - provider: val("--provider") ?? "any", - maxDepth: val("--depth") ? Number(val("--depth")) : undefined, - }); - } catch (e) { - console.error(` ${e.message}`); - process.exitCode = 1; - return; - } - if (json) return console.log(JSON.stringify(rec, null, 2)); - if (!rec.ok) { - console.error(` ${rec.reason}`); - process.exitCode = 1; - return; - } - heading( - `${BRAND.brand} route universal — ${rec.objective.kind}${rec.target != null ? ` (target ${rec.target.toFixed(2)})` : ""}\n`, - ); - rec.cascade.forEach((c, i) => { - console.log( - ` ${i === 0 ? "→" : "then, if a check fails →"} ${paint(c.model, "accent")} P(solve alone) ${c.pSolveAlone.toFixed(2)} · ~$${c.expectedAttemptCost.toFixed(3)}/attempt${c.status === "cold" ? " · cold (no outcomes yet)" : ""}`, - ); - }); - console.log( - `\n P(success) ${rec.pSuccess.toFixed(2)} · expected cost $${rec.expectedCost.toFixed(3)} · best single: ${rec.bestSingle.model} ${rec.bestSingle.pSuccess.toFixed(2)} at $${rec.bestSingle.expectedCost.toFixed(3)}`, - ); - console.log( - ` ${rec.candidates} candidate model(s), ${rec.cascadesEvaluated} cascade(s) compared · fit: ${rec.fit.origin}`, - ); - console.log( - ` learn from results: \`${BRAND.cli} route outcome "" --model --pass|--fail --cost \`, then \`${BRAND.cli} route fit\``, - ); -} - HANDLERS.anchor = async (argv) => { const { goalDrift, renderAnchor } = await import("./anchor.js"); const { clearGoal, getGoal, setGoal } = await import("./goal.js"); @@ -2495,90 +1475,6 @@ HANDLERS.diagnose = async (argv) => { } return; // advisory — halting the retry loop is the AGENT's move, not an exit code }; -HANDLERS.imagine = async (argv) => { - const { dryRun, imagineTask, renderImagine } = await import("./imagine.js"); - const json = argv.includes("--json"); - const doRun = argv.includes("--run"); - const allowDirty = argv.includes("--allow-dirty"); - const FLAGS = new Set(["--json", "--run", "--allow-dirty"]); - const task = argv - .slice(1) - .filter((a) => !FLAGS.has(a)) - .join(" "); - if (!task) { - console.error('usage: forge imagine "" [--run] [--allow-dirty] [--json]'); - process.exitCode = 1; - return; - } - const root = process.cwd(); - const r = imagineTask(root, task); - if (!doRun) { - console.log(json ? JSON.stringify(r, null, 2) : renderImagine(r)); - return; - } - // --run: the static prediction first (always), then the measured half. The sandbox - // is a git worktree of HEAD — uncommitted changes are INVISIBLE to it — so a dirty - // tree is refused by default rather than silently dry-running the wrong code. - if (!json) console.log(renderImagine(r, { footer: false })); - if (!allowDirty) { - let dirty = ""; - try { - const { execFileSync } = await import("node:child_process"); - dirty = execFileSync("git", ["status", "--porcelain"], { - cwd: root, - encoding: "utf8", - stdio: ["ignore", "pipe", "ignore"], - }).trim(); - } catch {} // not a repo / no git → dryRun reports its own precondition failure - if (dirty) { - console.error( - "\n imagine --run refused: the working tree is dirty and the sandbox runs HEAD,\n" + - " so your uncommitted changes would NOT be in the dry-run. Commit or stash them,\n" + - " or pass --allow-dirty to knowingly measure the last commit instead.", - ); - process.exitCode = 1; - return; - } - } - const d = dryRun(root, { tests: r.tests }); - // Metrics are best-effort telemetry (05-cost-model.md) — never let recording - // failure break the verdict. Only a run that happened is worth counting. - try { - if (d.durationMs !== undefined) { - const { record } = await import("./metrics.js"); - record(root, { - stage: "imagine", - outcome: d.ok && d.failed === 0 ? "clean" : "breaks", - ref: task.slice(0, 120), - durationMs: d.durationMs, - }); - } - } catch {} - if (json) { - console.log(JSON.stringify({ ...r, dryRun: d }, null, 2)); - return; - } - if (!d.ok) { - console.log(`\n dry-run: did not produce a verdict — ${d.reason}`); - if (d.output) console.log(`\n${d.output.replace(/^/gm, " ")}`); - process.exitCode = 1; - return; - } - console.log(`\n dry-run (sandboxed worktree of HEAD · ${d.runner}):`); - console.log( - ` pass ${d.passed} · fail ${d.failed} · ${d.durationMs}ms · worktree ${d.worktree}`, - ); - if (d.perFile) - for (const [t, s] of Object.entries(d.perFile)) - console.log(` ${s === "pass" ? "ok " : "FAIL"} ${t}`); - if (d.failed > 0) { - console.log("\n measured consequence: the selected suite BREAKS at HEAD — output tail:"); - console.log(`\n${(d.output || "").replace(/^/gm, " ")}`); - } else { - console.log("\n measured consequence: the selected suite is green at HEAD."); - } - return; -}; HANDLERS.lean = async (argv) => { const { leanRepo, renderLean } = await import("./lean.js"); const json = argv.includes("--json"); @@ -2634,349 +1530,6 @@ HANDLERS.scope = async (argv) => { return; }; /** The last line of a `uicheck design|visual` run, per overall verdict. */ -const VERDICT_LABEL = { - pass: "✓ PASS", - fail: "✗ FAIL", - "insufficient-signal": "✗ INSUFFICIENT SIGNAL — nothing measurable, so this is not a PASS", -}; -HANDLERS.uicheck = async (argv) => { - const sub = argv[1]; - if (sub === "visual") { - // The Playwright visual loop (spec §5): render in a real browser, fingerprint - // the COMPUTED styles, run the same design gate. Playwright is an optional - // tier (ADR-0005) — its absence is a note and exit 0, never a failure. - const { visualGate } = await import("./uivisual.js"); - const args = argv.slice(2); - const tasteIdx = args.indexOf("--taste"); - const tasteArg = tasteIdx >= 0 ? (args.splice(tasteIdx, 2)[1] ?? null) : null; - const json = args.includes("--json"); - const remote = args.includes("--remote"); - const targets = args.filter((a) => !a.startsWith("--")); - if (targets.length !== 1 || (tasteIdx >= 0 && !tasteArg)) { - console.error( - `usage: ${BRAND.cli} uicheck visual [--taste ] [--json] [--remote]`, - ); - process.exitCode = 1; - return; - } - const r = await visualGate(targets[0], { - taste: tasteArg, - remote, - root: process.cwd(), - }); - if (!r.ok) { - const reason = "reason" in r ? r.reason : "visual gate failed"; - if ("skipped" in r && r.skipped) { - // Graceful absence (ADR-0005): a missing optional tier is not a failure. - if (json) console.log(JSON.stringify({ skipped: true, reason }, null, 2)); - else { - heading(`${BRAND.brand} uicheck visual — skipped (no browser runtime)\n`); - console.log(` ${reason}`); - console.log( - " enable it: npm i -D playwright-core (or point FORGE_PLAYWRIGHT at an existing install, e.g. FORGE_PLAYWRIGHT=/path/to/node_modules/playwright-core)", - ); - } - return; // exit 0 — the static gate still stands - } - console.error(reason); - process.exitCode = 1; - return; - } - // The completion gate's UI evidence (a skipped run never reaches here, so a missing - // browser can never count as a PASS). - const { recordUiCheck } = await import("./gate.js"); - recordUiCheck(process.cwd(), { check: "visual", pass: !r.fail }); - if (json) { - const { ok: _ok, fail: _fail, ...body } = r; - console.log(JSON.stringify(body, null, 2)); - } else { - heading(`${BRAND.brand} uicheck visual — rendered fingerprint + design gate\n`); - console.log(` rendered: ${r.url} (${r.elements} visible element style(s))`); - console.log(` screenshots: ${r.screenshots.join(", ")}`); - if (r.taste) console.log(` taste: ${r.taste} (thresholds from its profile)`); - console.log( - ` slop distance: ${r.slop} (need ≥ ${r.tauSlop} — farther from generic is better)`, - ); - console.log( - r.hasProjectFingerprint - ? ` conformance: ${r.conform} (need ≤ ${r.tauConform} — closer to the project system is better)` - : ` conformance: (no project fingerprint claim — slop-only; mint one: \`${BRAND.cli} uicheck fingerprint --mint\`)`, - ); - for (const v of r.violations) console.log(`\n ✗ ${v.detail}\n fix: ${v.hint}`); - console.log(""); - for (const c of r.checks) - console.log( - ` ${c.pass ? "✓" : "✗"} ${c.id}: ${c.detail}${c.pass || !c.hint ? "" : `\n fix: ${c.hint}`}`, - ); - console.log(`\n ${VERDICT_LABEL[r.verdict]}`); - } - if (r.fail) process.exitCode = 1; - return; - } - if (sub === "interact") { - // The Playwright interaction loop (ROADMAP "Next"): drive the page and check what - // it DOES — keyboard reach, a visible focus ring, console cleanliness, reduced- - // motion honesty. Advisory by default (the `behavioral` oracle is cross-family- - // gated); --enforce or FORGE_ENFORCE=1 turns a fail into a non-zero exit. Playwright - // is an optional tier (ADR-0005) — its absence is a note and exit 0, never a failure. - const { runInteractions, recordInteraction } = await import("./uiinteract.js"); - const args = argv.slice(2); - const json = args.includes("--json"); - const remote = args.includes("--remote"); - const record = args.includes("--record"); - const enforce = args.includes("--enforce") || process.env.FORGE_ENFORCE === "1"; - const targets = args.filter((a) => !a.startsWith("--")); - if (targets.length !== 1) { - console.error( - `usage: ${BRAND.cli} uicheck interact [--record] [--enforce] [--json] [--remote]`, - ); - process.exitCode = 1; - return; - } - const r = await runInteractions(targets[0], { - remote, - cwd: process.cwd(), - }); - if (!r.ok) { - const reason = "reason" in r ? r.reason : "interaction run failed"; - if ("skipped" in r && r.skipped) { - // Graceful absence (ADR-0005): a missing optional tier is not a failure. - if (json) console.log(JSON.stringify({ skipped: true, reason }, null, 2)); - else { - heading(`${BRAND.brand} uicheck interact — skipped (no browser runtime)\n`); - console.log(` ${reason}`); - console.log( - " enable it: npm i -D playwright-core (or point FORGE_PLAYWRIGHT at an existing install)", - ); - } - return; // exit 0 — the advisory tier is absent - } - console.error(reason); - process.exitCode = 1; - return; - } - const recorded = record ? recordInteraction(process.cwd(), r.url, r.verdict) : null; - if (json) { - console.log(JSON.stringify({ url: r.url, ...r.verdict, recorded }, null, 2)); - } else { - heading(`${BRAND.brand} uicheck interact — browser interaction checks\n`); - console.log(` driven: ${r.url} (headless, prefers-reduced-motion)`); - for (const c of r.verdict.checks) console.log(` ${c.ok ? "✓" : "✗"} ${c.id}: ${c.detail}`); - console.log(`\n ${r.verdict.pass ? "✓ PASS" : "✗ FAIL"}${enforce ? "" : " (advisory)"}`); - if (recorded) - console.log( - recorded.recorded - ? ` recorded as behavioral evidence on design claim ${recorded.claimId.slice(0, 12)}` - : ` not recorded: ${recorded.reason}`, - ); - } - if (!r.verdict.pass && enforce) process.exitCode = 1; - return; - } - if (sub === "fingerprint" || sub === "design") { - const ui = await import("./uifingerprint.js"); - // `--taste ` (design only) takes a VALUE — splice it out before the - // file filter so the profile name is never mistaken for a file. - const args = argv.slice(2); - const tasteIdx = args.indexOf("--taste"); - const tasteArg = tasteIdx >= 0 ? (args.splice(tasteIdx, 2)[1] ?? null) : null; - // `--theme ` (repeatable) names the Tailwind theme sources explicitly; - // without it they are discovered (@theme / @tailwind stylesheets, tailwind.config.*). - /** @type {string[]} */ - const themeArgs = []; - let themeMissing = false; - for (let i = args.indexOf("--theme"); i >= 0; i = args.indexOf("--theme")) { - const [, value] = args.splice(i, 2); - if (!value || value.startsWith("--")) themeMissing = true; - else themeArgs.push(value); - } - const json = args.includes("--json"); - const files = args.filter((a) => !a.startsWith("--")); - if (!files.length || (tasteIdx >= 0 && !tasteArg) || themeMissing) { - console.error( - `usage: ${BRAND.cli} uicheck ${sub} [--theme ]... [--json]${sub === "fingerprint" ? " [--mint]" : " [--taste ]"}`, - ); - process.exitCode = 1; - return; - } - // An explicitly named theme that isn't there is an error, not a silent no-op. - const { existsSync } = await import("node:fs"); - const { resolve } = await import("node:path"); - const absent = themeArgs.filter((t) => !existsSync(resolve(process.cwd(), t))); - if (absent.length) { - console.error(`theme source not found: ${absent.join(", ")}`); - process.exitCode = 1; - return; - } - const theme = ui.loadThemeTokens( - process.cwd(), - ui.themeSourcesFor(process.cwd(), files, themeArgs), - ); - const themeLine = theme.sources.length - ? `${theme.sources.join(", ")} (${theme.colors.size} color · ${theme.radius.size} radius · ${theme.shadow.size} shadow token(s))` - : "(none found — token utilities like rounded-card / bg-brand stay unresolved; name one with --theme )"; - const themeSummary = { - sources: theme.sources, - colors: theme.colors.size, - radius: theme.radius.size, - shadow: theme.shadow.size, - }; - const fp = ui.fingerprintFiles(process.cwd(), files, { theme }); - if (sub === "fingerprint") { - let minted = null; - if (argv.includes("--mint")) { - const { epochDay } = await import("./util.js"); - minted = ui.mintProjectFingerprint(process.cwd(), files, { - t: epochDay(), - theme, - }); - } - if (json) { - console.log(JSON.stringify(minted ? { fingerprint: fp, minted } : fp, null, 2)); - } else { - heading(`${BRAND.brand} uicheck fingerprint — the design feature vector\n`); - console.log( - ` palette: ${fp.paletteSize} color(s), hue bins [${fp.hueBuckets.join(" ")}]`, - ); - console.log( - ` spacing: ${fp.spacing.join(", ") || "(none)"} px — base ${fp.spacingBase ?? "(none)"}, ${Math.round(fp.spacingOnScale * 100)}% on-scale`, - ); - console.log(` type: ${fp.fontFamilies.join(", ") || "(none)"}`); - console.log( - ` shape: radii ${fp.radii.join(", ") || "(none)"} (${fp.radiusLevels} level(s)) · ${fp.shadowLevels} shadow level(s)`, - ); - console.log(` theme: ${themeLine}`); - if (!ui.hasDesignSignal(fp)) - console.log( - "\n ! no measurable design feature in these files — `design` reports insufficient-signal", - ); - if (minted) { - if (minted.ok) - console.log( - `\n minted fingerprint claim ${minted.id.slice(0, 12)}${minted.existed ? " (already in ledger)" : ""} — the gate's "home"`, - ); - else console.error(`\n mint failed: ${"reason" in minted ? minted.reason : ""}`); - } - } - if (minted && !minted.ok) process.exitCode = 1; - return; - } - // design — the two-sided gate: fail when too close to generic OR (when the - // project has minted its fingerprint) too far from the project's own system. - // A taste profile (explicit --taste, else the style pinned by a - // `forge taste`-managed DESIGN.md) overrides thresholds + adds its checks. - const tasteName = tasteArg ?? ui.activeTasteStyle(process.cwd()); - const profile = tasteName ? ui.loadTasteProfile(tasteName) : null; - if (tasteArg && !profile) { - // Explicit --taste must exist; an auto-picked style without a JSON sibling - // silently falls back to defaults (custom prose styles stay legal). - console.error( - `unknown taste profile "${tasteArg}" — run \`${BRAND.cli} taste\` to list styles`, - ); - process.exitCode = 1; - return; - } - const projectFp = ui.loadProjectFingerprint(process.cwd()); - const tauSlop = profile?.gate?.tau_slop ?? ui.UI_GATE_DEFAULTS.tauSlop; - const tauConform = profile?.gate?.tau_conform ?? ui.UI_GATE_DEFAULTS.tauConform; - const gate = ui.uiGate(fp, { projectFp, tauSlop, tauConform }); - const checks = [...ui.scaleChecks(fp), ...(profile ? ui.profileChecks(fp, profile) : [])]; - // insufficient-signal (an empty vector) exits non-zero like FAIL: nothing was - // measured, so nothing passed. - const verdict = ui.overallVerdict(gate, checks); - // The completion gate's UI evidence: this verdict, bound to the current code state. - // Only a real PASS counts; an empty measurement is not evidence. - const { recordUiCheck } = await import("./gate.js"); - recordUiCheck(process.cwd(), { check: "design", pass: verdict === "pass", files }); - if (json) { - console.log( - JSON.stringify( - { - ...gate, - verdict, - checks, - hasProjectFingerprint: !!projectFp, - taste: profile ? tasteName : null, - tauSlop, - tauConform, - theme: themeSummary, - }, - null, - 2, - ), - ); - } else { - heading(`${BRAND.brand} uicheck design — slop distance + project conformance\n`); - if (profile) console.log(` taste: ${tasteName} (thresholds from its profile)`); - console.log(` theme: ${themeLine}`); - if (verdict !== "insufficient-signal") { - console.log( - ` slop distance: ${gate.slop} (need ≥ ${tauSlop} — farther from generic is better)`, - ); - console.log( - projectFp - ? ` conformance: ${gate.conform} (need ≤ ${tauConform} — closer to the project system is better)` - : ` conformance: (no project fingerprint claim — slop-only; mint one: \`${BRAND.cli} uicheck fingerprint --mint\`)`, - ); - } - for (const v of gate.violations) console.log(`\n ✗ ${v.detail}\n fix: ${v.hint}`); - if (verdict !== "insufficient-signal") { - console.log(""); - for (const c of checks) - console.log( - ` ${c.pass ? "✓" : "✗"} ${c.id}: ${c.detail}${c.pass || !c.hint ? "" : `\n fix: ${c.hint}`}`, - ); - } - console.log(`\n ${VERDICT_LABEL[verdict]}`); - } - if (verdict !== "pass") process.exitCode = 1; - return; - } - const { contrastReport, ASSERTABLE_CHECKS, ADVISORY_ONLY } = await import("./uicheck.js"); - // `uicheck contrast ` is the named form; bare `uicheck ` stays - // supported (it predates the subcommands and hooks already call it). Both exit 1 - // when the pair fails AA — a failing contrast must fail the script that asked. - const args = argv.slice(sub === "contrast" ? 2 : 1); - const json = args.includes("--json"); - const large = args.includes("--large"); - const colors = args.filter((a) => !a.startsWith("--")); - if (sub === "contrast" && colors.length !== 2) { - console.error( - `usage: ${BRAND.cli} uicheck contrast [--large] [--json] (colors: #hex[alpha], rgb(), hsl(), oklch(), oklab())`, - ); - process.exitCode = 1; - return; - } - const [fg, bg] = colors; - /** @type {ReturnType|null} */ - let r = null; - if (fg && bg) { - try { - r = contrastReport(fg, bg, { large }); - } catch (e) { - if (json) console.log(JSON.stringify({ error: e.message }, null, 2)); - else console.error(` ${e.message}`); - process.exitCode = 1; - return; - } - if (!r.passesAA) process.exitCode = 1; - if (json) { - console.log(JSON.stringify(r, null, 2)); - return; - } - } - heading(`${BRAND.brand} uicheck — deterministic UI review\n`); - if (r) { - const kind = large ? "large text / UI" : "normal text"; - console.log( - ` contrast ${fg} on ${bg}: ${r.ratio}:1 → ${r.level}${r.passesAA ? ` (passes AA for ${kind})` : ` (FAILS AA — ${kind} needs ${r.required.aa}:1)`}`, - ); - for (const n of r.notes) console.log(` note: ${n}`); - } - console.log(`\n ASSERT (deterministic): ${ASSERTABLE_CHECKS.map((c) => c.id).join(", ")}`); - console.log(` ADVISE (subjective, human-only): ${ADVISORY_ONLY.slice(0, 4).join(", ")} …`); - return; -}; HANDLERS.report = async (argv) => { // Static twin of `dash`: emit ONE self-contained HTML file (no server, opens // offline) instead of serving a live localhost lens. `--out ` overrides the diff --git a/src/cli/memory.js b/src/cli/memory.js new file mode 100644 index 00000000..30c1849f --- /dev/null +++ b/src/cli/memory.js @@ -0,0 +1,611 @@ +// forge CLI — the evidence-referenced memory commands — `ledger`, `reuse`, `context`. Moved verbatim out of src/cli.js (review A03: command dispatch +// and presentation are separated from domain operations; each domain's handlers live in one +// module, and the domain logic stays in the modules they import). cli.js registers these into +// its dispatch table; nothing here runs at import time. +import { BRAND, bar, heading, paint, table } from "./shared.js"; + +/** @type {Record unknown>} */ +const HANDLERS = {}; + +HANDLERS.ledger = async (argv) => { + const ls = await import("../ledger_store.js"); + const { epochDay, gitAuthor } = await import("../util.js"); + const root = process.cwd(); + // --personal targets the ledger beside the global recall store (~/.forge/recall/ + // ledger) — otherwise facts shadowed by `forge recall add` would be write-only, + // with no command able to inspect or verify them. + const personal = argv.includes("--personal"); + const args = argv.filter((a) => a !== "--json" && a !== "--personal"); + const dir = personal + ? (await import("node:path")).join((await import("../recall.js")).defaultStore(), "ledger") + : ls.repoLedger(root); + const sub = args[1] || "stats"; + const json = argv.includes("--json"); + const nowDay = epochDay(); + if (sub === "stats") { + const s = ls.stats(dir, nowDay); + if (json) return console.log(JSON.stringify(s, null, 2)); + heading(`${BRAND.brand} ledger — proof-carrying memory\n`); + console.log(` claims: ${s.total} (tombstoned ${s.tombstoned})`); + for (const [kind, n] of Object.entries(s.byKind)) console.log(` ${kind}: ${n}`); + if (s.pendingRetractions) + console.log( + paint( + ` ${s.pendingRetractions} claim(s) with an agent-proposed retraction — review, then \`forge ledger retract --reason …\``, + "warn", + ), + ); + console.log( + ` val: ${paint(`trusted ${s.val.trusted}`, "ok")} · ${paint(`uncertain ${s.val.uncertain}`, "warn")} · ${paint(`dormant ${s.val.dormant}`, "dim")}`, + ); + console.log( + paint("\n stored in .forge/ledger/ (git-committable, conflict-free merge)", "dim"), + ); + return; + } + if (sub === "verify") { + // --fix re-addresses claims still stored under their pre-CRLF-fold id. Reads accept + // that address either way, so this is not a repair — it is what stops one fact living + // at two addresses once a teammate on another platform mints its current form. + // `--fix --dry-run` previews the migration without writing (A11). + const migration = args.includes("--fix") + ? ls.migrateAddresses(dir, { dryRun: argv.includes("--dry-run") }) + : null; + const r = ls.verify(dir); + if (json) return console.log(JSON.stringify(migration ? { ...r, migration } : r, null, 2)); + if (migration) { + const { migrated, merged, failed } = migration; + console.log( + ` ${migration.dryRun ? "(dry run) would migrate" : "migrated"} ${migrated.length} claim(s) to their current address, ${migration.dryRun ? "would merge" : "merged"} ${merged.length} into an existing twin${failed.length ? `, ${failed.length} failed` : ""}`, + ); + } + console.log(` ${r.ok ? "OK" : "ISSUES"} — ${r.claims} claim(s), ${r.outcomes} outcome(s)`); + for (const i of r.issues) console.log(` - ${i}`); + if (!r.ok) process.exitCode = 1; + return; + } + if (sub === "show") { + const id = args[2]; + const hit = id && id.length >= 2 ? ls.getClaimByPrefix(dir, id, { attic: true }) : null; + if (!hit) { + console.error( + id ? ` no claim matching ${id}` : "usage: forge ledger show ", + ); + process.exitCode = 1; + return; + } + const { val } = await import("../ledger.js"); + const pending = ls.retractionProposals(ls.loadClaims(dir)).get(hit.id); + return console.log( + JSON.stringify( + { ...hit, val: val(hit, nowDay), ...(pending ? { pendingRetractions: pending } : {}) }, + null, + 2, + ), + ); + } + if (sub === "merge") { + const src = args[2]; + const { existsSync } = await import("node:fs"); + if (!src || !existsSync(src)) { + console.error( + src + ? ` no ledger at ${src}` + : "usage: forge ledger merge (a teammate's checkout, a backup, a worktree)", + ); + process.exitCode = 1; + return; + } + const r = ls.mergeDirs(dir, src); + if (json) return console.log(JSON.stringify(r, null, 2)); + console.log(` merged: ${r.claims} new claim(s), ${r.records} new record(s) — conflict-free`); + if (r.quarantined) + console.log( + ` quarantined: ${r.quarantined} invalid record(s) (forged hash or unresolvable ref — see quarantine/ in the ledger dir)`, + ); + return; + } + if (sub === "blame") { + const b = args[2] && args[2].length >= 2 ? ls.blame(dir, args[2], nowDay) : null; + if (!b) { + console.error( + args[2] ? ` no claim matching ${args[2]}` : "usage: forge ledger blame ", + ); + process.exitCode = 1; + return; + } + if (json) return console.log(JSON.stringify(b, null, 2)); + heading(`${BRAND.brand} ledger blame — ${b.kind} ${b.id.slice(0, 12)}\n`); + console.log( + ` val ${bar(b.val)} ${b.val.toFixed(2)} (trust-weighted ${b.valTrustWeighted.toFixed(2)})`, + ); + for (const p of b.minted) + console.log( + ` minted day ${p.t} by ${p.author || "(unknown)"}${p.agent ? ` · ${p.agent}` : ""}`, + ); + for (const e of b.evidence) + console.log( + ` ${e.result === "confirm" ? paint("confirm ", "ok") : paint("contradic", "err")} day ${e.t} ${e.oracle} → ${e.ref}${e.author ? ` by ${e.author}` : ""}`, + ); + for (const t of b.tombstones) + console.log( + paint(` retract day ${t.t} ${t.reason}${t.author ? ` by ${t.author}` : ""}`, "dim"), + ); + const trusts = Object.entries(b.trust); + if (trusts.length) { + console.log("\n author trust (earned from oracle outcomes on their claims):"); + for (const [a, u] of trusts) console.log(` ${u.toFixed(2)} ${a}`); + } + return; + } + // The two writes (08-dashboard-ux.md §2) — CLI twins of the dashboard's POSTs, so + // the dashboard stays a convenience, never a requirement. Both append-only. + if (sub === "ratify") { + const id = args[2]; + if (!id || id.length < 2) { + console.error("usage: forge ledger ratify "); + process.exitCode = 1; + return; + } + // Human-only promotion: the author is YOUR git identity, minted as a decision claim. + const r = ls.ratify(dir, id, { author: gitAuthor(), t: nowDay }); + if (!r.ok) { + console.error(` ${r.reason}`); + process.exitCode = 1; + return; + } + if (json) return console.log(JSON.stringify(r, null, 2)); + console.log( + ` ratified ${r.ratifies.slice(0, 12)} → decision ${r.decisionId.slice(0, 12)}${r.existed ? " (already ratified — same decision)" : ""}`, + ); + return; + } + if (sub === "retract") { + const id = args[2]; + const ri = args.indexOf("--reason"); + const reason = ri >= 0 ? (args[ri + 1] ?? "") : ""; + if (!id || id === "--reason" || !reason) { + console.error('usage: forge ledger retract --reason ""'); + process.exitCode = 1; + return; + } + // A tombstone is permanent, so it must name exactly one claim: the full 64-char id, + // never a prefix (a short prefix used to retract the first sorted match). + if (!ls.FULL_ID_RE.test(id)) { + console.error( + ` refused: retract needs the full 64-character claim id (got "${id}") — see \`forge ledger query\` or \`forge ledger show \``, + ); + process.exitCode = 1; + return; + } + const hit = ls.getClaimByPrefix(dir, id); + if (!hit) { + console.error(` no claim matching ${id}`); + process.exitCode = 1; + return; + } + const r = ls.tombstone(dir, hit.id, { + author: gitAuthor(), + reason, + t: nowDay, + }); + if (!r.ok) { + console.error(` ${r.reason}`); + process.exitCode = 1; + return; + } + if (json) return console.log(JSON.stringify({ ...r, id: hit.id }, null, 2)); + console.log( + ` retracted ${hit.id.slice(0, 12)} — ${reason}${r.deduped ? " (already retracted with this record)" : ""}`, + ); + return; + } + // `compact` — archive what this ledger's own history says will not be used again, and + // near-duplicates, printing every learned number (ledger_retention.js). Reversible. + if (sub === "compact") { + const dryRun = argv.includes("--dry-run"); + const r = ls.compactLedger(dir, nowDay, { dryRun }); + if (json) return console.log(JSON.stringify(r, null, 2)); + const rt = r.retention; + const d = r.duplicates; + const lines = [ + `Forge ledger — compact (every cut-off learned from this ledger)${dryRun ? " [dry run]" : ""}`, + "", + ` claims: ${r.claims} · claims with logged use: ${r.servedClaims}`, + rt.learned + ? ` retention: idle cut-off ${rt.cutoff} d = the longest idle stretch any claim came back from (${rt.comebacks} comebacks, typical gap ${rt.typicalGap} d; usage log spans ${rt.usageSpan} d)` + : ` retention: not learned — ${rt.reason}`, + d?.boundary != null + ? ` duplicates: boundary ${d.boundary.toFixed(2)} (two components beat one: BIC ${d.bic2?.toFixed(1)} < ${d.bic1?.toFixed(1)}) · ${d.groups.length} group(s)` + : ` duplicates: none — ${d?.compared ? `one component fits the ${d.compared} nearest-neighbour similarities better` : "fewer than two claims of one kind are still live to compare"}`, + "", + ` archive: ${r.archive.length}`, + ]; + for (const a of r.archive.slice(0, 20)) lines.push(` ${a.id.slice(0, 12)} ${a.reason}`); + if (r.archive.length > 20) lines.push(` … ${r.archive.length - 20} more (--json for all)`); + // Similar-but-opposite pairs are never archived as duplicates (review F16): a person decides. + const conflicts = d?.conflicts ?? []; + if (conflicts.length) { + lines.push( + "", + ` kept apart — similar but conflicting (review, then retract one): ${conflicts.length}`, + ); + for (const c of conflicts.slice(0, 10)) + lines.push(` ${c.a.slice(0, 12)} ↔ ${c.b.slice(0, 12)} ${c.conflicts}`); + } + lines.push( + "", + dryRun + ? " dry run: nothing written" + : ` archived ${r.archived.length} claim(s) to .forge/ledger/attic/ — new evidence brings one back; show/blame still read it`, + ); + return console.log(lines.join("\n")); + } + if (sub === "query") { + const q = args.slice(2).join(" "); + if (!q) { + console.error('usage: forge ledger query ""'); + process.exitCode = 1; + return; + } + const { retrieve, claimText } = await import("../ledger.js"); + // The embeddings tier (ADR-0005) is assembled HERE, not in ledger.js — the pure + // core stays provider-free. No FORGE_EMBED (or a failing provider) → sim is + // null and retrieval is the stock MinHash path. + const { claimSim, simLabel } = await import("../embed.js"); + const claims = ls.loadClaims(dir); + const sim = claimSim(root, q, claims, claimText); + const ranked = retrieve(q, claims, { nowDay, budget: 8, sim }); + ls.recordUse( + dir, + ranked.map((r) => r.claim.id), + { via: "cli.query", t: nowDay }, + ); + if (json) + return console.log( + JSON.stringify( + { + sim: simLabel(sim), + results: ranked.map((r) => ({ + id: r.claim.id, + kind: r.claim.kind, + score: r.score, + })), + }, + null, + 2, + ), + ); + console.log(paint(` sim: ${simLabel(sim)}`, "dim")); + if (!ranked.length) return console.log(" no matching live claims"); + for (const r of ranked) + console.log( + ` ${bar(r.score, 8)} ${r.score.toFixed(3)} ${paint(r.claim.kind.padEnd(9), "accent")} ${paint(r.claim.id.slice(0, 8), "dim")} ${claimText(r.claim).slice(0, 90)}`, + ); + return; + } + // `at` / `diff` / `root` — the temporal surface. The store is append-only and every + // record carries its day, so a past day's beliefs are recomputed, never guessed. + const parseDay = (s) => { + if (/^\d{1,6}$/.test(s ?? "")) return Number(s); // bare epoch-day + const t = Date.parse(`${s}T00:00:00Z`); + if (Number.isNaN(t)) return null; + // Round-trip check: Date.parse silently rolls impossible dates over (2026-02-31 + // → March 3rd), which would answer a temporal query for a day nobody asked about. + if (new Date(t).toISOString().slice(0, 10) !== s) return null; + return Math.floor(t / 86_400_000); + }; + if (sub === "at") { + const day = parseDay(args[2]); + if (day === null) { + console.error(`usage: ${BRAND.cli} ledger at [--json]`); + process.exitCode = 1; + return; + } + const lg = await import("../ledger.js"); + const live = lg.liveClaims(lg.stateAt(ls.loadState(dir), day)); + const rows = live + .map((c) => ({ + id: c.id, + kind: c.kind, + val: Number(lg.val(c, day).toFixed(4)), + tombstoned: Boolean(c.tombstone), + text: lg.claimText(c).slice(0, 90), + })) + .sort((a, b) => b.val - a.val || (a.id < b.id ? -1 : 1)); + if (json) return console.log(JSON.stringify({ day, claims: rows.length, rows }, null, 2)); + heading(`${BRAND.brand} ledger — beliefs as of day ${day}\n`); + const byKind = {}; + for (const r of rows) byKind[r.kind] = (byKind[r.kind] ?? 0) + 1; + console.log( + ` claims: ${rows.length} (${rows.filter((r) => r.tombstoned).length} tombstoned)`, + ); + for (const [kind, n] of Object.entries(byKind)) console.log(` ${kind}: ${n}`); + for (const r of rows.slice(0, 10)) + console.log( + ` ${bar(r.val, 8)} ${r.val.toFixed(3)} ${paint(r.kind.padEnd(9), "accent")} ${paint(r.id.slice(0, 8), "dim")} ${r.text}`, + ); + return; + } + if (sub === "diff") { + const a = parseDay(args[2]); + const b = args[3] ? parseDay(args[3]) : nowDay; + if (a === null || b === null) { + console.error( + `usage: ${BRAND.cli} ledger diff [] [--json]`, + ); + process.exitCode = 1; + return; + } + if (a > b) { + // beliefDiff's contract is dayA ≤ dayB; a reversed window would print silently + // inverted appeared/retired classes, so refuse loudly instead. + console.error(` (day ${a}) is after (day ${b}) — swap the arguments`); + process.exitCode = 1; + return; + } + const lg = await import("../ledger.js"); + const d = lg.beliefDiff(ls.loadState(dir), a, b); + if (json) return console.log(JSON.stringify({ since: a, until: b, ...d }, null, 2)); + heading(`${BRAND.brand} ledger — what changed, day ${a} → ${b}\n`); + console.log( + ` appeared ${d.appeared.length} · retired ${d.retired.length} · ${paint(`strengthened ${d.strengthened.length}`, "ok")} · ${paint(`weakened ${d.weakened.length}`, "warn")}`, + ); + const row = (label, r) => + console.log( + ` ${label} ${paint(r.kind.padEnd(9), "accent")} ${paint(r.id.slice(0, 8), "dim")} ${r.from === null ? "· " : r.from.toFixed(2)} → ${r.to === null ? "·" : r.to.toFixed(2)} ${r.text.slice(0, 70)}`, + ); + for (const r of d.appeared.slice(0, 5)) row(paint("new ", "ok"), r); + for (const r of d.retired.slice(0, 5)) row(paint("gone", "err"), r); + for (const r of d.strengthened.slice(0, 5)) row(paint("up ", "ok"), r); + for (const r of d.weakened.slice(0, 5)) row(paint("down", "warn"), r); + return; + } + if (sub === "root") { + const lg = await import("../ledger.js"); + const r = lg.stateRoot(ls.loadState(dir)); + if (json) return console.log(JSON.stringify(r, null, 2)); + console.log(r.root); // bare hex on stdout — scriptable ("are we in sync?" is one diff) + return; + } + if (sub === "sync") { + const { ledgerSync, defaultRef } = await import("../ledger_sync.js"); + const di = args.indexOf("--dir"); + const re = args.indexOf("--remote"); + const rf = args.indexOf("--ref"); + const r = ledgerSync({ + dir, + root, + personal, + dirTarget: di >= 0 ? args[di + 1] : undefined, + remote: re >= 0 ? args[re + 1] : undefined, + ref: rf >= 0 ? args[rf + 1] : undefined, + }); + if (json) return console.log(JSON.stringify(r, null, 2)); + heading(`${BRAND.brand} ledger sync\n`); + if (!r.ok) { + console.error(` ${paint(r.reason ?? "sync failed", "err")}`); + process.exitCode = 1; + return; + } + if (r.mode === "dir") { + console.log( + table([ + [paint("target", "dim"), r.dir], + [paint("pulled", "dim"), `${r.pulled.claims} claim(s), ${r.pulled.records} record(s)`], + [ + paint("pushed", "dim"), + r.upToDate + ? paint("up to date — state roots match, nothing to merge", "dim") + : `${r.pushed.claims} claim(s), ${r.pushed.records} record(s)`, + ], + ]), + ); + } else { + console.log( + table([ + [paint("ref", "dim"), `${r.remote} ${r.ref}`], + [paint("pulled", "dim"), `${r.pulled.claims} claim(s), ${r.pulled.records} record(s)`], + [ + paint("pushed", "dim"), + r.upToDate + ? paint("up to date — nothing to push", "dim") + : `yes (retries ${r.retries})`, + ], + ]), + ); + } + for (const n of r.notes ?? []) console.log(paint(` note: ${n}`, "warn")); + if (r.mode === "ref" && r.ref === defaultRef(personal)) + console.log(paint("\n synced through a git ref — CRDT, converges in any order", "dim")); + return; + } + if (sub === "import") { + const b = await import("../ledger_bridge.js"); + let r; + if (personal) { + // Personal import: facts from the global recall store into the personal ledger. + const { defaultStore } = await import("../recall.js"); + r = { + lessons: 0, + outcomes: 0, + ...b.importFacts(defaultStore(), dir, nowDay), + }; + } else { + const { brainStore } = await import("../brain.js"); + r = b.importLegacy(root, { + recallStore: brainStore(root), + recallLedger: dir, + nowDay, + }); + } + if (json) return console.log(JSON.stringify(r, null, 2)); + console.log(` imported: ${r.lessons} lesson(s), ${r.facts} fact(s), ${r.outcomes} outcome(s)`); + for (const x of r.refused) console.log(` refused: ${x}`); + return; + } + console.error( + `ledger: unknown subcommand "${sub}" (stats | verify | show | blame | query | at | diff [] | root | ratify | retract --reason "" | merge | sync [--dir |--remote |--ref ] | import) [--personal] [--json]`, + ); + process.exitCode = 1; + return; +}; + +HANDLERS.reuse = async (argv) => { + const ru = await import("../reuse.js"); + const { load: loadAtlas } = await import("../atlas.js"); + const { epochDay } = await import("../util.js"); + const root = process.cwd(); + const nowDay = epochDay(); + const json = argv.includes("--json"); + const flagVal = (name) => { + const i = argv.indexOf(name); + return i >= 0 ? argv[i + 1] : undefined; + }; + const args = argv.filter( + (a, i) => !a.startsWith("--") && argv[i - 1] !== "--file" && argv[i - 1] !== "--ref", + ); + const sub = args[1] || "stats"; + if (sub === "query") { + const spec = args.slice(2).join(" "); + if (!spec) { + console.error('usage: forge reuse query "" [--json]'); + process.exitCode = 1; + return; + } + const r = ru.reuseQuery(root, spec, { atlas: loadAtlas(root), nowDay }); + if (json) + return console.log( + JSON.stringify( + { + tier: r.tier, + artifact: r.artifact?.id, + jaccard: r.jaccard, + similarity: r.similarity, + sim: r.sim, + revalidation: r.revalidation?.status, + requiresRevalidation: r.requiresRevalidation === true, + reasons: r.reasons, + }, + null, + 2, + ), + ); + console.log(` sim: ${r.sim}`); + if (r.tier === "miss") { + console.log(" miss — nothing verified matches; generate, then `forge reuse mint` it"); + } else { + const a = r.artifact; + console.log( + ` ${r.tier.toUpperCase()} hit (similarity ${(r.similarity ?? r.jaccard ?? 1).toFixed(2)}) — ${a.body.form}${a.body.code?.path ? ` at ${a.body.code.path}` : ""}`, + ); + console.log( + ` claim ${a.id.slice(0, 12)} — \`forge ledger blame ${a.id.slice(0, 8)}\` for its proof`, + ); + if (r.tier === "near") + console.log(" near tier: a reworded match — review the diff before reusing it as-is"); + if (r.tier === "adapt") + console.log(" adapt tier: inject as a verified starting point, generate only the delta"); + if (r.requiresRevalidation) + console.log( + ` NOT revalidated: ${(r.revalidation?.unknown ?? []).join(", ")} — check before use`, + ); + } + for (const why of r.reasons) console.log(` note: ${why}`); + return; + } + if (sub === "mint") { + const spec = args.slice(2).join(" "); + const file = flagVal("--file"); + const ref = flagVal("--ref"); + if (!spec || !file) { + console.error( + 'usage: forge reuse mint "" --file [--ref ] [--json]', + ); + process.exitCode = 1; + return; + } + const { repoLedger } = await import("../ledger_store.js"); + // With an atlas, each dependency's declaration is fingerprinted, so a later signature + // change invalidates the artifact (review F05). + const desc = ru.describeFile(root, file, { atlas: loadAtlas(root) }); + const r = ru.mintArtifact( + repoLedger(root), + { spec, form: "module", ...desc }, + ref + ? { + evidence: { oracle: "test.run", result: "confirm", ref }, + t: nowDay, + } + : { t: nowDay }, + ); + if (json) return console.log(JSON.stringify(r, null, 2)); + if (!r.ok) { + console.error(` ${r.reason}`); + process.exitCode = 1; + return; + } + console.log( + ` minted: ${r.id.slice(0, 12)} (${desc.iface.length} export(s), ${desc.deps.length} dep(s))`, + ); + console.log( + r.serves + ? " serving: yes — verification evidence attached" + : " serving: NOT YET — no evidence; attach a verified test/commit ref (--ref) or it stays at the 0.5 prior", + ); + return; + } + if (sub === "stats") { + const { summarize } = await import("../metrics.js"); + const s = summarize(root).cache ?? { + events: 0, + byOutcome: {}, + savedEstimate: 0, + }; + if (json) return console.log(JSON.stringify(s, null, 2)); + heading(`${BRAND.brand} reuse — proof-carrying code cache\n`); + console.log(` lookups: ${s.events}`); + for (const [o, n] of Object.entries(s.byOutcome)) console.log(` ${o}: ${n}`); + console.log(` est. tokens saved: ${s.savedEstimate}`); + return; + } + console.error( + `reuse: unknown subcommand "${sub}" (query | mint --file | stats)`, + ); + process.exitCode = 1; + return; +}; + +HANDLERS.context = async (argv) => { + const { assemble, renderContext } = await import("../context.js"); + const { load: loadAtlas } = await import("../atlas.js"); + const { epochDay } = await import("../util.js"); + const json = argv.includes("--json"); + const bi = argv.indexOf("--budget"); + const budget = bi >= 0 ? Number(argv[bi + 1]) || undefined : undefined; + const task = argv + .filter((a, i) => i > 0 && !a.startsWith("--") && argv[i - 1] !== "--budget") + .join(" "); + if (!task) { + console.error('usage: forge context "" [--budget ] [--json]'); + process.exitCode = 1; + return; + } + const r = assemble(process.cwd(), task, { + atlas: loadAtlas(process.cwd()), + nowDay: epochDay(), + ...(budget ? { budget } : {}), + }); + // --block delivers the assembled context itself — the spans the summary talks about (R13). + const withBlock = argv.includes("--block"); + if (json) { + const { block, ...rest } = r; + console.log(JSON.stringify(withBlock ? r : rest, null, 2)); + } else if (withBlock) { + console.log(r.block); + } else console.log(renderContext(r)); + if (!r.ok) process.exitCode = 1; + return; +}; + +export default HANDLERS; diff --git a/src/cli/routing.js b/src/cli/routing.js new file mode 100644 index 00000000..02d92004 --- /dev/null +++ b/src/cli/routing.js @@ -0,0 +1,381 @@ +// forge CLI — the routing and cost commands — `route` (incl. `route universal`), `models`, `cost`. Moved verbatim out of src/cli.js (review A03: command dispatch +// and presentation are separated from domain operations; each domain's handlers live in one +// module, and the domain logic stays in the modules they import). cli.js registers these into +// its dispatch table; nothing here runs at import time. +import { BRAND, bar, heading, paint } from "./shared.js"; + +/** @type {Record unknown>} */ +const HANDLERS = {}; + +HANDLERS.cost = async (argv) => { + // `--stages` is the P8 measured report (per-stage factors from .forge/metrics.jsonl); + // the default path stays the ccusage per-day spend view, untouched. + if (argv.includes("--stages")) { + const { renderCostReport, report } = await import("../cost_report.js"); + const r = report(process.cwd()); + console.log(argv.includes("--json") ? JSON.stringify(r, null, 2) : renderCostReport(r)); + return; + } + const { execFileSync } = await import("node:child_process"); + const run = (bin, args) => execFileSync(bin, args, { encoding: "utf8", stdio: "pipe" }); + heading(`${BRAND.brand} cost — real per-day spend (ccusage)\n`); + try { + let out; + try { + out = run("ccusage", ["daily"]); + } catch { + // Pinned (verified 2026-07-05) — never @latest for code we execute; re-verify via dev-radar. + out = run("npx", ["-y", "ccusage@20.0.14", "daily"]); + } + console.log(out.trim()); + } catch { + const { estimateSpendFromLogs } = await import("../cost_report.js"); + const est = estimateSpendFromLogs({ root: process.cwd() }); + if (est && est.totalCost > 0) { + console.log( + ` $${est.totalCost.toFixed(2)} estimated from Claude session logs (${est.sessions} session(s))`, + ); + if (est.byModel.length) { + for (const m of est.byModel) + console.log( + ` ${m.model.padEnd(30)} ${m.priced ? `$${m.cost.toFixed(4)}` : "unpriced"} (${m.inTokens} in / ${m.outTokens} out)${m.priceSource ? ` · price: ${m.priceSource}` : ""}`, + ); + } + if (est.unpriced?.length) + console.log( + paint( + ` not in the total — no catalog or snapshot price for: ${est.unpriced.join(", ")}`, + "dim", + ), + ); + console.log(paint("\n install ccusage for precise tracking: npm i -g ccusage", "dim")); + } else { + console.log( + " ccusage not found. Install for real spend (reads local JSONL, nothing leaves your machine):\n npm i -g ccusage # then: forge cost", + ); + } + } + console.log( + paint( + `\n ceiling: FORGE_COST_CEILING (default $10) — the cost-budget guard warns when a day exceeds it.`, + "dim", + ), + ); + return; +}; + +HANDLERS.models = async (argv) => { + // What each tier resolves to RIGHT NOW: family → newest model in the active provider's live + // catalog (else the shipped snapshot), priced from OpenRouter's catalog (else the snapshot). + const { describeResolution, PRICING_VERIFIED, resolveTiers } = await import("../model_tiers.js"); + const { activeProvider, envModelOverride } = await import("../providers.js"); + const root = process.cwd(); + const provider = activeProvider(root); + const tiers = resolveTiers({ root, provider }); + const override = envModelOverride(); + if (argv.includes("--json")) + return console.log( + JSON.stringify( + { provider: provider.name, override, pricingVerified: PRICING_VERIFIED, tiers }, + null, + 2, + ), + ); + heading(`${BRAND.brand} models — each tier's family, resolved to a concrete model\n`); + console.log(` provider ${provider.name} (${provider.label || provider.name})`); + if (override) + console.log( + ` override ${override} — ANTHROPIC_MODEL/FORGE_MODEL pins every call; the tiers below apply without it`, + ); + const priceText = (p) => (p ? `$${p.inCost}/$${p.outCost}` : "—"); + const priceFrom = (p) => + !p + ? "unpriced" + : p.source === "catalog" + ? "catalog" + : `snapshot${p.basis === "family" ? " (tier)" : ""}`; + const width = Math.max(28, ...tiers.map((t) => (t.model?.id ?? "").length + 2)); + console.log( + `\n ${"tier".padEnd(8)} ${"family".padEnd(7)} ${"model".padEnd(width)} ${"created".padEnd(11)} ${"$/M tok".padEnd(10)} ${"id from".padEnd(9)} price from`, + ); + for (const t of tiers) { + console.log( + ` ${t.class.padEnd(8)} ${t.family.padEnd(7)} ${(t.model?.id ?? "—").padEnd(width)} ${(t.model?.createdAt?.slice(0, 10) ?? "—").padEnd(11)} ${priceText(t.price).padEnd(10)} ${(t.model?.source ?? "—").padEnd(9)} ${priceFrom(t.price)}`, + ); + } + console.log(""); + for (const t of tiers) console.log(` ${t.family.padEnd(7)} ${describeResolution(t.model)}`); + const live = tiers.find((t) => t.price?.source === "catalog")?.price; + console.log( + live + ? `\n prices: live from ${live.catalog}${live.cache === "stale" ? " (last cached copy — catalog unreachable)" : ""}` + : `\n prices: shipped snapshot, verified ${PRICING_VERIFIED} (OpenRouter's catalog unavailable or unlisted)`, + ); + console.log( + paint( + " cache: .forge/cache/ — reused while the response's own Cache-Control/Expires says fresh, else revalidated (ETag); FORGE_NO_CATALOG_FETCH=1 stays offline", + "dim", + ), + ); + return; +}; + +HANDLERS.route = async (argv) => { + const r = await import("../route.js"); + if (argv[1] === "gateway") { + const result = r.emitGatewayConfig(process.cwd()); + if (typeof result === "object" && !result.ok) { + console.log(` ${result.reason}`); + return; + } + console.log(` wrote ${result} — LiteLLM tiers: forge-simple / forge-medium / forge-complex.`); + console.log(" next: pin+install litellm, run it, point ANTHROPIC_BASE_URL at it, then"); + console.log( + " REQUEST the tier `forge route` recommends (a plain claude-* request passes through).", + ); + return; + } + if (argv[1] === "calibrate") { + // Advisory → gated promotion (ROADMAP): measure whether an affine calibration of the + // routing rubric beats the raw rubric on a held-out split of the HAND-LABELLED fixture + // (there is no outcome data). Advisory — routing keeps the rubric unless the gate + // promotes AND a caller adopts the calibration, which nothing in src/ does. + const res = r.calibrateRouting(); + if (argv.includes("--json")) return console.log(JSON.stringify(res, null, 2)); + heading(`${BRAND.brand} route calibrate — rubric calibration check (measured gate)\n`); + console.log(` samples: ${res.n} hand-labelled task phrase(s) — no routing outcomes exist`); + if (res.baselineMetric !== undefined) + console.log( + ` held-out MAE: rubric ${res.baselineMetric} · calibrated ${res.candidateMetric}`, + ); + console.log( + res.mode === "candidate" + ? ` → PROMOTE calibration — ${res.reason} (a=${res.model.a.toFixed(3)}, b=${res.model.b.toFixed(3)})` + : ` → keep the rubric — ${res.reason}`, + ); + console.log( + "\n advisory — routing stays on the rubric; nothing adopts a promoted calibration yet,", + ); + console.log(" and calibrating on real routing outcomes needs data forge does not record"); + return; + } + if (["universal", "outcome", "fit", "models"].includes(argv[1]) || argv.includes("--universal")) { + return routeUniversalCli(argv); + } + const json = argv.includes("--json"); + const apply = argv.includes("--apply"); + const providerIdx = argv.indexOf("--provider"); + const providerName = providerIdx >= 0 ? argv[providerIdx + 1] : undefined; + const FLAGS = new Set(["--json", "--apply"]); + const task = argv + .slice(1) + .filter((a, i) => !FLAGS.has(a) && a !== "--provider" && argv[i] !== "--provider") + .join(" "); + if (!task) { + console.error( + 'usage: forge route "" [--apply] [--provider ] [--json] | forge route gateway', + ); + process.exitCode = 1; + return; + } + if (providerName) { + const { setProvider } = await import("../providers.js"); + const sr = setProvider(process.cwd(), providerName); + if (!sr.ok) { + console.error(` ${sr.reason}`); + process.exitCode = 1; + return; + } + } + const rec = r.routeTask(process.cwd(), task); + r.meterRoute(process.cwd(), task, rec); + // The recommendation is a tier (a model family); its concrete id and price are resolved here, + // in the command — routeTask stays network-free because the hooks run it. BOTH output modes + // resolve, so a script reading --json never sees a different model than the text prints. + const { describeResolution, resolveTierModel, resolveTierPrice } = await import( + "../model_tiers.js" + ); + const { activeProvider } = await import("../providers.js"); + const opts = { root: process.cwd(), provider: activeProvider(process.cwd()) }; + const resolved = resolveTierModel(rec.key, opts); + const price = resolveTierPrice(rec.key, { ...opts, resolved }); + if (json) { + // `model` stays the snapshot row (its shape is public); `resolved` is what would be called. + console.log(JSON.stringify({ ...rec, resolved, price }, null, 2)); + } else { + heading(`${BRAND.brand} route — cheapest capable model\n`); + const name = + resolved?.source === "catalog" && resolved.displayName + ? resolved.displayName + : rec.model.name; + console.log( + ` → ${paint(name, "accent")} (${rec.tier}, ${price ? `${price.inCost}/${price.outCost} per M tok, ${price.source === "catalog" ? "live price" : "current effective"}` : "price unknown"})`, + ); + if (resolved) console.log(` model: ${resolved.id} — ${describeResolution(resolved)}`); + console.log(` ${rec.model.use}`); + console.log( + ` complexity ${bar(rec.score, 8)} ${rec.score.toFixed(2)}${rec.reasons.length ? ` · driven by: ${rec.reasons.join(", ")}` : ""}`, + ); + console.log( + ` signals: ${rec.signals.files} file(s), fan-out ${rec.signals.fanout}, churn ${rec.signals.churn}, past-mistakes ${rec.signals.pastMistakes}, ambiguity ${rec.signals.ambiguity.toFixed(2)}`, + ); + } + if (apply) { + const { applyRoute } = await import("../providers.js"); + const ar = applyRoute(rec.key); + if (ar.ok) { + if (!json) console.log(`\n applied: model set to ${ar.model} (${ar.modelId}) in ${ar.path}`); + } else { + if (!json) console.error(`\n apply failed: ${ar.reason}`); + process.exitCode = 1; + } + } else if (!json) { + console.log( + `\n advisory · apply: \`${BRAND.cli} route "" --apply\` · gateway: \`${BRAND.cli} route gateway\``, + ); + } + return; +}; +// Universal router (src/router): any provider's models, chosen by expected cost for the success +// probability asked for. Models come from data/models.json and .forge/models.json. + +// Universal router (src/router): any provider's models, chosen by expected cost for the success +// probability asked for. Models come from data/models.json and .forge/models.json. +async function routeUniversalCli(argv) { + const U = await import("../router/index.js"); + const { loadRegistry } = await import("../router/registry.js"); + const json = argv.includes("--json"); + const val = (flag) => (argv.includes(flag) ? argv[argv.indexOf(flag) + 1] : undefined); + const VALUED = new Set([ + "--objective", + "--provider", + "--model", + "--cost", + "--depth", + "--attempt", + "--verify-run", + ]); + const words = argv + .slice(1) + .filter((a, i, arr) => !a.startsWith("--") && !VALUED.has(arr[i - 1] ?? "")); + const sub = ["outcome", "fit", "models", "universal"].includes(words[0]) + ? words.shift() + : "universal"; + const root = process.cwd(); + if (sub === "models") { + const reg = loadRegistry(root); + const fit = U.loadRouterModel(root); + const rows = reg.models.map((m) => ({ + id: m.id, + org: m.org ?? null, + status: fit?.models.includes(m.id) ? "fitted" : "cold", + providers: Object.keys(m.providers ?? {}), + price: m.price_in != null ? `${m.price_in}/${m.price_out}` : null, + })); + if (json) + return console.log( + JSON.stringify({ sources: reg.sources, fit: fit?.origin ?? null, models: rows }, null, 2), + ); + heading(`${BRAND.brand} route models — registry (${reg.sources.join(" + ")})\n`); + for (const r of rows) + console.log( + ` ${r.id.padEnd(22)} ${String(r.org ?? "").padEnd(16)} ${r.status.padEnd(7)} ${r.price ? `$${r.price}/Mtok`.padEnd(14) : "".padEnd(14)} ${r.providers.join(", ") || "(no provider id: advice only)"}`, + ); + console.log(`\n fit in use: ${fit?.origin ?? "none"}`); + return; + } + if (sub === "fit") { + const model = U.fitRouter(root); + if (json) return console.log(JSON.stringify(model.provenance, null, 2)); + console.log( + ` refit on ${model.provenance.local.outcomes} recorded outcome(s) over ${model.provenance.local.tasks} task(s); wrote .forge/router_model.json`, + ); + return; + } + const task = words.join(" "); + if (!task) { + console.error( + 'usage: forge route universal "" [--objective match-best-single|target:

    |value:<$>|budget:<$>] [--provider |any] [--depth ] [--json]\n' + + ' forge route outcome "" --model --pass|--fail [--cost ] [--attempt ] [--verify-run ]\n' + + " forge route fit | forge route models", + ); + process.exitCode = 1; + return; + } + if (sub === "outcome") { + const passed = argv.includes("--pass") ? true : argv.includes("--fail") ? false : undefined; + const cost = val("--cost") !== undefined ? Number(val("--cost")) : null; + try { + const row = U.recordOutcome(root, { + task, + model: val("--model"), + passed, + cost, + attemptId: val("--attempt") ?? null, + verifyRunId: val("--verify-run") ?? null, + }); + if (json) return console.log(JSON.stringify(row, null, 2)); + console.log( + row.duplicate + ? ` attempt ${row.attemptId} was already recorded — not counted twice` + : ` recorded ${row.model} ${row.passed ? "pass" : "fail"} (${row.provenance}) for task ${row.task} (.forge/route_outcomes.jsonl)`, + ); + } catch (e) { + console.error(` ${e.message}`); + process.exitCode = 1; + } + return; + } + let rec; + try { + rec = U.routeUniversal(root, task, { + objective: val("--objective"), + provider: val("--provider") ?? "any", + maxDepth: val("--depth") ? Number(val("--depth")) : undefined, + }); + } catch (e) { + console.error(` ${e.message}`); + process.exitCode = 1; + return; + } + if (json) { + console.log(JSON.stringify(rec, null, 2)); + if (!rec.ok) process.exitCode = 1; + return; + } + if (!rec.ok) { + console.error(` ${rec.feasible === false ? "INFEASIBLE — " : ""}${rec.reason}`); + // F12: the least-bad cascade is shown only as an explicit, labeled fallback. + const fb = rec.fallback; + if (fb) + console.error( + ` fallback (does NOT meet the objective): ${fb.cascade.map((c) => c.model).join(" → ")} · P(success) ${fb.pSuccess.toFixed(2)} · expected $${fb.expectedCost.toFixed(3)} (up to $${fb.maxPossibleCost.toFixed(3)} if every attempt runs)`, + ); + process.exitCode = 1; + return; + } + heading( + `${BRAND.brand} route universal — ${rec.objective.kind}${rec.target != null ? ` (target ${rec.target.toFixed(2)})` : ""}\n`, + ); + rec.cascade.forEach((c, i) => { + console.log( + ` ${i === 0 ? "→" : "then, if a check fails →"} ${paint(c.model, "accent")} P(solve alone) ${c.pSolveAlone.toFixed(2)} · ~$${c.expectedAttemptCost.toFixed(3)}/attempt${c.status === "cold" ? " · cold (no outcomes yet)" : ""}${c.providers.length ? "" : " · no provider id"}`, + ); + }); + // A07: a model in the registry is not a model you can call. + if (!rec.applicable) + console.log( + `\n ${paint("advice only", "warn")}: ${rec.unmapped.join(", ")} ${rec.unmapped.length === 1 ? "has" : "have"} no provider id — add one under "providers" in .forge/models.json, or pass --provider to route among models you can call (\`${BRAND.cli} route models\` lists who serves what)`, + ); + console.log( + `\n P(success) ${rec.pSuccess.toFixed(2)} · expected cost $${rec.expectedCost.toFixed(3)} (not a cap; up to $${rec.maxPossibleCost.toFixed(3)} if every attempt runs) · best single: ${rec.bestSingle.model} ${rec.bestSingle.pSuccess.toFixed(2)} at $${rec.bestSingle.expectedCost.toFixed(3)}`, + ); + console.log( + ` ${rec.candidates} candidate model(s), ${rec.cascadesEvaluated} cascade(s) compared · fit: ${rec.fit.origin}`, + ); + console.log( + ` learn from results: \`${BRAND.cli} route outcome "" --model --pass|--fail --cost \`, then \`${BRAND.cli} route fit\``, + ); +} + +export default HANDLERS; diff --git a/src/cli/shared.js b/src/cli/shared.js new file mode 100644 index 00000000..a7091bb0 --- /dev/null +++ b/src/cli/shared.js @@ -0,0 +1,16 @@ +// forge CLI — presentation helpers shared by every command module (src/cli.js and +// src/cli/*.js), defined once so the handler modules cannot drift apart. +import { BRAND } from "../brand.js"; +// Color is capability-gated (FORCE_COLOR > NO_COLOR > TERM=dumb > TTY) — piped +// output stays byte-plain, so nothing downstream ever parses an escape code. +import { bar, heading as fmtHeading, paint, table } from "../fmt.js"; + +export { BRAND, bar, paint, table }; + +// Per-command title lines ("Forge — …") are branding chrome, not results. They +// print only when asked (`--verbose` or FORGE_VERBOSE=1); by default a command emits +// just its output. The `--help`/`--version` banner is unaffected. +export const VERBOSE = process.argv.includes("--verbose") || process.env.FORGE_VERBOSE === "1"; +export const heading = (text) => { + if (VERBOSE) console.log(fmtHeading(text)); +}; diff --git a/src/cli/verification.js b/src/cli/verification.js new file mode 100644 index 00000000..d5b8a490 --- /dev/null +++ b/src/cli/verification.js @@ -0,0 +1,567 @@ +// forge CLI — the verification commands — `verify`, `imagine`, `uicheck`. Moved verbatim out of src/cli.js (review A03: command dispatch +// and presentation are separated from domain operations; each domain's handlers live in one +// module, and the domain logic stays in the modules they import). cli.js registers these into +// its dispatch table; nothing here runs at import time. +import { BRAND, bar, heading, paint, table } from "./shared.js"; + +/** @type {Record unknown>} */ +const HANDLERS = {}; + +HANDLERS.verify = async (argv) => { + const json = argv.includes("--json"); + if (argv.includes("--deep")) { + const { verifyDeep, LENSES } = await import("../consensus.js"); + // `--llm` opts the reviewer panel in for this run; otherwise FORGE_LLM decides. + const r = verifyDeep({ + targetRoot: process.cwd(), + llm: argv.includes("--llm") ? true : undefined, + }); + if (json) { + console.log(JSON.stringify(r, null, 2)); + if (!r.ok) process.exitCode = 1; + return; + } + heading(`${BRAND.brand} verify --deep — multi-lens consensus\n`); + console.log( + table( + r.lenses.map((l) => { + const meta = LENSES[l.lens]; + const state = + l.ran === false + ? paint("— skipped", "dim") + : l.s > 0 + ? paint("● finding", meta.solo ? "err" : "warn") + : paint("✓ clean", "ok"); + return [l.lens, meta.family, `w=${meta.weight}`, state]; + }), + ), + ); + if (r.findings.length) { + console.log(); + for (const f of r.findings) console.log(` ! ${f}`); + } + console.log( + `\n defectRiskScore: ${bar(r.p)} ${r.p.toFixed(2)} (heuristic, not a calibrated probability)${ + r.families.length ? ` (families: ${r.families.join(", ")})` : "" + }`, + ); + console.log( + ` remainingUncheckedWeight: ${r.residual.toFixed(3)} — Theorem-D silent-miss bound (heuristic)`, + ); + // The core tests state, spelled out — deep ok REQUIRES a passing core, so the + // reader must see whether a verifier actually ran (RA-01). + const t = r.tests ?? /** @type {import("../verify.js").VerifyTests} */ ({ ran: false }); + console.log( + ` tests: ${t.status ?? "unknown"}${t.runner ? ` (${t.runner})` : ""}${ + !t.runner && t.detected?.length ? ` (detected: ${t.detected.join(", ")})` : "" + }`, + ); + if (t.executed?.length) + console.log( + ` suites ran: ${t.executed.map((s) => `${s.label}=${s.status}`).join(", ")}`, + ); + if (t.notExecuted?.length) + console.log(` suites skipped: ${t.notExecuted.join(", ")} (no built-in executor)`); + const detail = t.output || (t.detected ?? []).join(", "); + const verdictLine = + r.status === "PASS" + ? paint("PASS", "ok") + : r.status === "NOT_CONFIGURED" + ? paint("NOT VERIFIED — no test runner configured (NOT_CONFIGURED)", "warn") + : r.status === "INCOMPLETE" + ? paint(`NOT VERIFIED — tests incomplete${detail ? ` (${detail})` : ""}`, "warn") + : paint("BLOCKED — cross-family consensus says defect", "err"); + console.log(`\n ${verdictLine}`); + if (!r.ok) process.exitCode = 1; + return; + } + const { verify } = await import("../verify.js"); + const r = verify({ targetRoot: process.cwd() }); + if (json) { + console.log(JSON.stringify(r, null, 2)); + if (!r.ok) process.exitCode = 1; + return; + } + heading(`${BRAND.brand} verify\n`); + console.log(` changed files: ${r.changedFiles.length}`); + // Honest four-state tests line (RA-09): "nothing ran" is never dressed up as a pass, + // and a real run names the runner that actually executed. + const t = r.tests; + const testsLine = + t.status === "PASS" + ? `✓ pass (${t.runner ?? "project suite"})` + : t.status === "FAIL" + ? `✗ FAIL (${t.runner ?? "project suite"})` + : t.status === "INCOMPLETE" + ? `— INCOMPLETE: ${t.output || (t.detected ?? []).join(", ") || "test run did not complete"}` + : "— NOT CONFIGURED (no test runner detected)"; + console.log(` tests: ${testsLine}`); + // Per-suite honesty (HI-01): a polyglot repo runs every executable suite; name which + // ones ran with what verdict, and which were detected but never executed. + if (t.executed?.length) + console.log( + ` suites ran: ${t.executed.map((s) => `${s.label}=${s.status}`).join(", ")}`, + ); + if (t.notExecuted?.length) + console.log(` suites skipped: ${t.notExecuted.join(", ")} (no built-in executor)`); + // Coverage (review F08): which package dirs the verdict actually speaks for. + const cov = t.coverage; + if (cov && cov.required.length > 1) + console.log( + ` packages: ${cov.covered.length}/${cov.required.length} covered${ + cov.uncovered.length ? ` — no verdict for ${cov.uncovered.join(", ")}` : "" + }${cov.excluded.length ? ` (${cov.excluded.length} excluded)` : ""}`, + ); + if (t.mutated) + console.log(" ! the code changed while the tests ran — the verdict is not bound to it"); + console.log(` symbols checked: ${r.provenance.symbolsChecked}`); + if (r.unknown.length) + console.log( + ` ! not in codebase (possible hallucination): ${r.unknown.slice(0, 12).join(", ")}`, + ); + console.log(` provenance: .forge/provenance.json (run ${r.provenance.event?.runId})`); + // BLOCKED is reserved for a runner that actually FAILED; anything that never ran + // to completion is NOT VERIFIED (still exit 1 — unverified is not a pass). + const verdict = r.ok + ? "PASS" + : t.status === "FAIL" + ? "BLOCKED — tests failing" + : `NOT VERIFIED — ${t.status}`; + console.log(`\n ${verdict}`); + if (!r.ok) process.exitCode = 1; + return; +}; + +HANDLERS.imagine = async (argv) => { + const { dryRun, imagineTask, renderImagine } = await import("../imagine.js"); + const json = argv.includes("--json"); + const doRun = argv.includes("--run"); + const allowDirty = argv.includes("--allow-dirty"); + const FLAGS = new Set(["--json", "--run", "--allow-dirty"]); + const task = argv + .slice(1) + .filter((a) => !FLAGS.has(a)) + .join(" "); + if (!task) { + console.error('usage: forge imagine "" [--run] [--allow-dirty] [--json]'); + process.exitCode = 1; + return; + } + const root = process.cwd(); + const r = imagineTask(root, task); + if (!doRun) { + console.log(json ? JSON.stringify(r, null, 2) : renderImagine(r)); + return; + } + // --run: the static prediction first (always), then the measured half. The sandbox + // is a git worktree of HEAD — uncommitted changes are INVISIBLE to it — so a dirty + // tree is refused by default rather than silently dry-running the wrong code. + if (!json) console.log(renderImagine(r, { footer: false })); + if (!allowDirty) { + let dirty = ""; + try { + const { execFileSync } = await import("node:child_process"); + dirty = execFileSync("git", ["status", "--porcelain"], { + cwd: root, + encoding: "utf8", + stdio: ["ignore", "pipe", "ignore"], + }).trim(); + } catch {} // not a repo / no git → dryRun reports its own precondition failure + if (dirty) { + console.error( + "\n imagine --run refused: the working tree is dirty and the isolated checkout runs HEAD,\n" + + " so your uncommitted changes would NOT be in the dry-run. Commit or stash them,\n" + + " or pass --allow-dirty to knowingly measure the last commit instead.", + ); + process.exitCode = 1; + return; + } + } + const d = dryRun(root, { tests: r.tests }); + // Metrics are best-effort telemetry (05-cost-model.md) — never let recording + // failure break the verdict. Only a run that happened is worth counting. + try { + if (d.durationMs !== undefined) { + const { record } = await import("../metrics.js"); + record(root, { + stage: "imagine", + outcome: d.ok && d.failed === 0 ? "clean" : "breaks", + ref: task.slice(0, 120), + durationMs: d.durationMs, + }); + } + } catch {} + if (json) { + console.log(JSON.stringify({ ...r, dryRun: d }, null, 2)); + return; + } + if (!d.ok) { + console.log(`\n dry-run: did not produce a verdict — ${d.reason}`); + if (d.output) console.log(`\n${d.output.replace(/^/gm, " ")}`); + process.exitCode = 1; + return; + } + console.log( + `\n dry-run (isolated checkout of HEAD — not a security sandbox · ${d.runner ?? "node --test"}):`, + ); + console.log( + ` pass ${d.passed} · fail ${d.failed} · ${d.durationMs}ms · worktree ${d.worktree}`, + ); + if (d.perFile) + for (const [t, s] of Object.entries(d.perFile)) + console.log(` ${s === "pass" ? "ok " : "FAIL"} ${t}`); + if (d.failed > 0) { + console.log("\n measured consequence: the selected suite BREAKS at HEAD — output tail:"); + console.log(`\n${(d.output || "").replace(/^/gm, " ")}`); + } else { + console.log("\n measured consequence: the selected suite is green at HEAD."); + } + return; +}; + +const VERDICT_LABEL = { + pass: "✓ PASS", + fail: "✗ FAIL", + "insufficient-signal": "✗ INSUFFICIENT SIGNAL — nothing measurable, so this is not a PASS", +}; + +HANDLERS.uicheck = async (argv) => { + const sub = argv[1]; + if (sub === "visual") { + // The Playwright visual loop (spec §5): render in a real browser, fingerprint + // the COMPUTED styles, run the same design gate. Playwright is an optional + // tier (ADR-0005) — its absence is a note and exit 0, never a failure. + const { visualGate } = await import("../uivisual.js"); + const args = argv.slice(2); + const tasteIdx = args.indexOf("--taste"); + const tasteArg = tasteIdx >= 0 ? (args.splice(tasteIdx, 2)[1] ?? null) : null; + const json = args.includes("--json"); + const remote = args.includes("--remote"); + const targets = args.filter((a) => !a.startsWith("--")); + if (targets.length !== 1 || (tasteIdx >= 0 && !tasteArg)) { + console.error( + `usage: ${BRAND.cli} uicheck visual [--taste ] [--json] [--remote]`, + ); + process.exitCode = 1; + return; + } + const r = await visualGate(targets[0], { + taste: tasteArg, + remote, + root: process.cwd(), + }); + if (!r.ok) { + const reason = "reason" in r ? r.reason : "visual gate failed"; + if ("skipped" in r && r.skipped) { + // Graceful absence (ADR-0005): a missing optional tier is not a failure. + if (json) console.log(JSON.stringify({ skipped: true, reason }, null, 2)); + else { + heading(`${BRAND.brand} uicheck visual — skipped (no browser runtime)\n`); + console.log(` ${reason}`); + console.log( + " enable it: npm i -D playwright-core (or point FORGE_PLAYWRIGHT at an existing install, e.g. FORGE_PLAYWRIGHT=/path/to/node_modules/playwright-core)", + ); + } + return; // exit 0 — the static gate still stands + } + console.error(reason); + process.exitCode = 1; + return; + } + // The completion gate's UI evidence (a skipped run never reaches here, so a missing + // browser can never count as a PASS). + const { recordUiCheck } = await import("../gate.js"); + recordUiCheck(process.cwd(), { check: "visual", pass: !r.fail }); + if (json) { + const { ok: _ok, fail: _fail, ...body } = r; + console.log(JSON.stringify(body, null, 2)); + } else { + heading(`${BRAND.brand} uicheck visual — rendered fingerprint + design gate\n`); + console.log(` rendered: ${r.url} (${r.elements} visible element style(s))`); + console.log(` screenshots: ${r.screenshots.join(", ")}`); + if (r.taste) console.log(` taste: ${r.taste} (thresholds from its profile)`); + console.log( + ` slop distance: ${r.slop} (need ≥ ${r.tauSlop} — farther from generic is better)`, + ); + console.log( + r.hasProjectFingerprint + ? ` conformance: ${r.conform} (need ≤ ${r.tauConform} — closer to the project system is better)` + : ` conformance: (no project fingerprint claim — slop-only; mint one: \`${BRAND.cli} uicheck fingerprint --mint\`)`, + ); + for (const v of r.violations) console.log(`\n ✗ ${v.detail}\n fix: ${v.hint}`); + console.log(""); + for (const c of r.checks) + console.log( + ` ${c.pass ? "✓" : "✗"} ${c.id}: ${c.detail}${c.pass || !c.hint ? "" : `\n fix: ${c.hint}`}`, + ); + console.log(`\n ${VERDICT_LABEL[r.verdict]}`); + } + if (r.fail) process.exitCode = 1; + return; + } + if (sub === "interact") { + // The Playwright interaction loop (ROADMAP "Next"): drive the page and check what + // it DOES — keyboard reach, a visible focus ring, console cleanliness, reduced- + // motion honesty. Advisory by default (the `behavioral` oracle is cross-family- + // gated); --enforce or FORGE_ENFORCE=1 turns a fail into a non-zero exit. Playwright + // is an optional tier (ADR-0005) — its absence is a note and exit 0, never a failure. + const { runInteractions, recordInteraction } = await import("../uiinteract.js"); + const args = argv.slice(2); + const json = args.includes("--json"); + const remote = args.includes("--remote"); + const record = args.includes("--record"); + const enforce = args.includes("--enforce") || process.env.FORGE_ENFORCE === "1"; + const targets = args.filter((a) => !a.startsWith("--")); + if (targets.length !== 1) { + console.error( + `usage: ${BRAND.cli} uicheck interact [--record] [--enforce] [--json] [--remote]`, + ); + process.exitCode = 1; + return; + } + const r = await runInteractions(targets[0], { + remote, + cwd: process.cwd(), + }); + if (!r.ok) { + const reason = "reason" in r ? r.reason : "interaction run failed"; + if ("skipped" in r && r.skipped) { + // Graceful absence (ADR-0005): a missing optional tier is not a failure. + if (json) console.log(JSON.stringify({ skipped: true, reason }, null, 2)); + else { + heading(`${BRAND.brand} uicheck interact — skipped (no browser runtime)\n`); + console.log(` ${reason}`); + console.log( + " enable it: npm i -D playwright-core (or point FORGE_PLAYWRIGHT at an existing install)", + ); + } + return; // exit 0 — the advisory tier is absent + } + console.error(reason); + process.exitCode = 1; + return; + } + const recorded = record ? recordInteraction(process.cwd(), r.url, r.verdict) : null; + if (json) { + console.log(JSON.stringify({ url: r.url, ...r.verdict, recorded }, null, 2)); + } else { + heading(`${BRAND.brand} uicheck interact — browser interaction checks\n`); + console.log(` driven: ${r.url} (headless, prefers-reduced-motion)`); + for (const c of r.verdict.checks) console.log(` ${c.ok ? "✓" : "✗"} ${c.id}: ${c.detail}`); + console.log(`\n ${r.verdict.pass ? "✓ PASS" : "✗ FAIL"}${enforce ? "" : " (advisory)"}`); + if (recorded) + console.log( + recorded.recorded + ? ` recorded as behavioral evidence on design claim ${recorded.claimId.slice(0, 12)}` + : ` not recorded: ${recorded.reason}`, + ); + } + if (!r.verdict.pass && enforce) process.exitCode = 1; + return; + } + if (sub === "fingerprint" || sub === "design") { + const ui = await import("../uifingerprint.js"); + // `--taste ` (design only) takes a VALUE — splice it out before the + // file filter so the profile name is never mistaken for a file. + const args = argv.slice(2); + const tasteIdx = args.indexOf("--taste"); + const tasteArg = tasteIdx >= 0 ? (args.splice(tasteIdx, 2)[1] ?? null) : null; + // `--theme ` (repeatable) names the Tailwind theme sources explicitly; + // without it they are discovered (@theme / @tailwind stylesheets, tailwind.config.*). + /** @type {string[]} */ + const themeArgs = []; + let themeMissing = false; + for (let i = args.indexOf("--theme"); i >= 0; i = args.indexOf("--theme")) { + const [, value] = args.splice(i, 2); + if (!value || value.startsWith("--")) themeMissing = true; + else themeArgs.push(value); + } + const json = args.includes("--json"); + const files = args.filter((a) => !a.startsWith("--")); + if (!files.length || (tasteIdx >= 0 && !tasteArg) || themeMissing) { + console.error( + `usage: ${BRAND.cli} uicheck ${sub} [--theme ]... [--json]${sub === "fingerprint" ? " [--mint]" : " [--taste ]"}`, + ); + process.exitCode = 1; + return; + } + // An explicitly named theme that isn't there is an error, not a silent no-op. + const { existsSync } = await import("node:fs"); + const { resolve } = await import("node:path"); + const absent = themeArgs.filter((t) => !existsSync(resolve(process.cwd(), t))); + if (absent.length) { + console.error(`theme source not found: ${absent.join(", ")}`); + process.exitCode = 1; + return; + } + const theme = ui.loadThemeTokens( + process.cwd(), + ui.themeSourcesFor(process.cwd(), files, themeArgs), + ); + const themeLine = theme.sources.length + ? `${theme.sources.join(", ")} (${theme.colors.size} color · ${theme.radius.size} radius · ${theme.shadow.size} shadow token(s))` + : "(none found — token utilities like rounded-card / bg-brand stay unresolved; name one with --theme )"; + const themeSummary = { + sources: theme.sources, + colors: theme.colors.size, + radius: theme.radius.size, + shadow: theme.shadow.size, + }; + const fp = ui.fingerprintFiles(process.cwd(), files, { theme }); + if (sub === "fingerprint") { + let minted = null; + if (argv.includes("--mint")) { + const { epochDay } = await import("../util.js"); + minted = ui.mintProjectFingerprint(process.cwd(), files, { + t: epochDay(), + theme, + }); + } + if (json) { + console.log(JSON.stringify(minted ? { fingerprint: fp, minted } : fp, null, 2)); + } else { + heading(`${BRAND.brand} uicheck fingerprint — the design feature vector\n`); + console.log( + ` palette: ${fp.paletteSize} color(s), hue bins [${fp.hueBuckets.join(" ")}]`, + ); + console.log( + ` spacing: ${fp.spacing.join(", ") || "(none)"} px — base ${fp.spacingBase ?? "(none)"}, ${Math.round(fp.spacingOnScale * 100)}% on-scale`, + ); + console.log(` type: ${fp.fontFamilies.join(", ") || "(none)"}`); + console.log( + ` shape: radii ${fp.radii.join(", ") || "(none)"} (${fp.radiusLevels} level(s)) · ${fp.shadowLevels} shadow level(s)`, + ); + console.log(` theme: ${themeLine}`); + if (!ui.hasDesignSignal(fp)) + console.log( + "\n ! no measurable design feature in these files — `design` reports insufficient-signal", + ); + if (minted) { + if (minted.ok) + console.log( + `\n minted fingerprint claim ${minted.id.slice(0, 12)}${minted.existed ? " (already in ledger)" : ""} — the gate's "home"`, + ); + else console.error(`\n mint failed: ${"reason" in minted ? minted.reason : ""}`); + } + } + if (minted && !minted.ok) process.exitCode = 1; + return; + } + // design — the two-sided gate: fail when too close to generic OR (when the + // project has minted its fingerprint) too far from the project's own system. + // A taste profile (explicit --taste, else the style pinned by a + // `forge taste`-managed DESIGN.md) overrides thresholds + adds its checks. + const tasteName = tasteArg ?? ui.activeTasteStyle(process.cwd()); + const profile = tasteName ? ui.loadTasteProfile(tasteName) : null; + if (tasteArg && !profile) { + // Explicit --taste must exist; an auto-picked style without a JSON sibling + // silently falls back to defaults (custom prose styles stay legal). + console.error( + `unknown taste profile "${tasteArg}" — run \`${BRAND.cli} taste\` to list styles`, + ); + process.exitCode = 1; + return; + } + const projectFp = ui.loadProjectFingerprint(process.cwd()); + const tauSlop = profile?.gate?.tau_slop ?? ui.UI_GATE_DEFAULTS.tauSlop; + const tauConform = profile?.gate?.tau_conform ?? ui.UI_GATE_DEFAULTS.tauConform; + const gate = ui.uiGate(fp, { projectFp, tauSlop, tauConform }); + const checks = [...ui.scaleChecks(fp), ...(profile ? ui.profileChecks(fp, profile) : [])]; + // insufficient-signal (an empty vector) exits non-zero like FAIL: nothing was + // measured, so nothing passed. + const verdict = ui.overallVerdict(gate, checks); + // The completion gate's UI evidence: this verdict, bound to the current code state. + // Only a real PASS counts; an empty measurement is not evidence. + const { recordUiCheck } = await import("../gate.js"); + recordUiCheck(process.cwd(), { check: "design", pass: verdict === "pass", files }); + if (json) { + console.log( + JSON.stringify( + { + ...gate, + verdict, + checks, + hasProjectFingerprint: !!projectFp, + taste: profile ? tasteName : null, + tauSlop, + tauConform, + theme: themeSummary, + }, + null, + 2, + ), + ); + } else { + heading(`${BRAND.brand} uicheck design — slop distance + project conformance\n`); + if (profile) console.log(` taste: ${tasteName} (thresholds from its profile)`); + console.log(` theme: ${themeLine}`); + if (verdict !== "insufficient-signal") { + console.log( + ` slop distance: ${gate.slop} (need ≥ ${tauSlop} — farther from generic is better)`, + ); + console.log( + projectFp + ? ` conformance: ${gate.conform} (need ≤ ${tauConform} — closer to the project system is better)` + : ` conformance: (no project fingerprint claim — slop-only; mint one: \`${BRAND.cli} uicheck fingerprint --mint\`)`, + ); + } + for (const v of gate.violations) console.log(`\n ✗ ${v.detail}\n fix: ${v.hint}`); + if (verdict !== "insufficient-signal") { + console.log(""); + for (const c of checks) + console.log( + ` ${c.pass ? "✓" : "✗"} ${c.id}: ${c.detail}${c.pass || !c.hint ? "" : `\n fix: ${c.hint}`}`, + ); + } + console.log(`\n ${VERDICT_LABEL[verdict]}`); + } + if (verdict !== "pass") process.exitCode = 1; + return; + } + const { contrastReport, ASSERTABLE_CHECKS, ADVISORY_ONLY } = await import("../uicheck.js"); + // `uicheck contrast ` is the named form; bare `uicheck ` stays + // supported (it predates the subcommands and hooks already call it). Both exit 1 + // when the pair fails AA — a failing contrast must fail the script that asked. + const args = argv.slice(sub === "contrast" ? 2 : 1); + const json = args.includes("--json"); + const large = args.includes("--large"); + const colors = args.filter((a) => !a.startsWith("--")); + if (sub === "contrast" && colors.length !== 2) { + console.error( + `usage: ${BRAND.cli} uicheck contrast [--large] [--json] (colors: #hex[alpha], rgb(), hsl(), oklch(), oklab())`, + ); + process.exitCode = 1; + return; + } + const [fg, bg] = colors; + /** @type {ReturnType|null} */ + let r = null; + if (fg && bg) { + try { + r = contrastReport(fg, bg, { large }); + } catch (e) { + if (json) console.log(JSON.stringify({ error: e.message }, null, 2)); + else console.error(` ${e.message}`); + process.exitCode = 1; + return; + } + if (!r.passesAA) process.exitCode = 1; + if (json) { + console.log(JSON.stringify(r, null, 2)); + return; + } + } + heading(`${BRAND.brand} uicheck — deterministic UI review\n`); + if (r) { + const kind = large ? "large text / UI" : "normal text"; + console.log( + ` contrast ${fg} on ${bg}: ${r.ratio}:1 → ${r.level}${r.passesAA ? ` (passes AA for ${kind})` : ` (FAILS AA — ${kind} needs ${r.required.aa}:1)`}`, + ); + for (const n of r.notes) console.log(` note: ${n}`); + } + console.log(`\n ASSERT (deterministic): ${ASSERTABLE_CHECKS.map((c) => c.id).join(", ")}`); + console.log(` ADVISE (subjective, human-only): ${ADVISORY_ONLY.slice(0, 4).join(", ")} …`); + return; +}; + +export default HANDLERS; diff --git a/src/commands.js b/src/commands.js index 8bce40f4..4d77ea7d 100644 --- a/src/commands.js +++ b/src/commands.js @@ -106,7 +106,17 @@ export const COMMANDS = { ledger: "evidence-referenced memory — stats / verify / show / blame / query / compact / at / diff / root / ratify / retract / merge / sync / import", reuse: "proof-carrying code cache — query / mint --file / stats", - context: "budgeted context assembly + completeness gate — what an edit NEEDS known", + context: { + summary: + "budgeted context assembly + completeness gate — what an edit NEEDS known, delivered or owed", + usage: 'forge context "" [--budget ] [--block] [--json]', + flags: [ + { flag: "--budget ", desc: "assembly budget (chars/3.6 estimate; default 6000)" }, + { flag: "--block", desc: "print the assembled context block itself (what gets delivered)" }, + { flag: "--json", desc: "machine-readable result (add --block to include the block)" }, + ], + examples: ['forge context "update computeTax in src/tax.js" --block'], + }, preflight: "assumption check — what a task names that the repo doesn't define", config: "provider setup — show / switch / add providers, set default model", route: diff --git a/src/context.js b/src/context.js index 2feac0f8..3aec0849 100644 --- a/src/context.js +++ b/src/context.js @@ -2,11 +2,13 @@ // gate (docs/plans/substrate-v2/04-context-assembly.md). Two failures die here: // over-stuffing (everything competes for the window on equal terms — P3 of the // paper) and under-supplying (the agent edits a symbol without its callers, tests, -// or the team's lessons, then "assumes"). Selection gets an objective function -// (greedy knapsack by value density, with a compression ladder instead of silent -// drops) and sufficiency becomes a COMPUTED SET: required knowledge R(edit) from -// the atlas, missing = R \ covered — auto-fetched when resolvable, asked as a -// derived M2 question when not. Context insufficiency stops being a feeling. +// or the team's lessons, then "assumes"). Selection gets an objective (a greedy +// value-density heuristic with a compression ladder instead of silent drops — no +// approximation guarantee is claimed) and sufficiency becomes a COMPUTED SET: required +// knowledge R(edit) from the atlas, missing = R \ covered — auto-fetched when resolvable, +// asked as a derived M2 question when not. "Covered" means DELIVERED in the block: a +// pointer to a file is a pending read, not coverage (review F03), and a block that cannot +// fit the budget says so instead of claiming completion (review F02). import { existsSync, readFileSync } from "node:fs"; import { basename, dirname, join } from "node:path"; import { has as atlasHas, query as atlasQuery, impact } from "./atlas.js"; @@ -14,8 +16,20 @@ import { claimText, val } from "./ledger.js"; import { loadClaims, repoLedger } from "./ledger_store.js"; import { referencedEntities } from "./preflight.js"; -/** chars → tokens heuristic (calibrated in P8; consistent with the reuse estimator). */ +/** chars → tokens ESTIMATE (chars/3.6, consistent with the reuse estimator). It is not a + * model tokenizer: a hard window limit needs the target model's own tokenizer, so treat + * every budget here as an estimate with that stated basis. */ export const tokensOf = (text) => Math.ceil(String(text).length / 3.6); +/** How `tokens` is measured — reported with every assembly so nobody mistakes it for exact. */ +export const TOKEN_ESTIMATE = "chars/3.6 estimate of the rendered block"; +/** Items are joined with this separator; its cost is budgeted like any other text. */ +const SEP = "\n\n"; +/** Direct dependents listed by name before the rest are summarized as omitted. */ +const DEPS_SHOWN = 12; +/** Lines of source shown after a definition's line in a symbol-span variant. */ +const SPAN_AFTER = 40; +const SPAN_BEFORE = 2; +const HEAD_LINES = 25; /** Lessons must be THIS trusted to enter the required set (spec §3: lessons*(S)). */ export const LESSON_REQUIRED_VAL = 0.8; @@ -100,23 +114,93 @@ export function requiredSet(root, task, { atlas = null, claims = [], nowDay = 0 // An item is one injectable unit with a COMPRESSION LADDER: granularity variants from // full text down to a one-line pointer. The optimizer may downgrade an item instead of -// dropping it — compression is a lossy move with a known cost, chosen explicitly, -// never by scroll-off (spec §2). -function fileItem(root, rel, { covers, source, score }) { +// dropping it — compression is a lossy move with a known cost, chosen explicitly, never by +// scroll-off (spec §2). EVERY VARIANT CARRIES ITS OWN COVERAGE (review F03): the full file +// covers all its keys; a symbol span covers the definitions whose declaration line it shows; +// the first-25-lines head covers only what lies inside it; a pointer (`- read `) +// covers NOTHING — it creates a pending read obligation. Availability is not delivery. + +/** One variant: its rendered text, estimated tokens, the keys it satisfies, the keys it only + * points at (pending reads), and what it truncated. */ +const variant = (gran, text, covers, pending = [], truncated = null) => ({ + gran, + text, + tokens: tokensOf(text), + covers, + pending, + ...(truncated ? { truncated } : {}), +}); + +/** + * All variants of one file, given the required keys it serves: `needs` entries are + * {key, kind: "def"|"file"|"tests", line?}. Ordered largest → smallest; a variant that is not + * smaller than the previous one is skipped. + */ +function fileItem(root, rel, { needs, source, score }) { const text = readRel(root, rel); if (text === null) return null; - const head = text.split("\n").slice(0, 25).join("\n"); - const variants = [ - { gran: "full", text: `// ${rel}\n${text}`, tokens: tokensOf(text) }, - { gran: "head", text: `// ${rel} (first 25 lines)\n${head}`, tokens: tokensOf(head) }, - { gran: "pointer", text: `- read ${rel}`, tokens: 8 }, - ]; - return { id: `${source}:${rel}`, source, covers, score, variants }; + const lines = text.split("\n"); + const total = lines.length; + const keys = needs.map((n) => n.key); + const defs = needs.filter((n) => n.kind === "def" && Number.isFinite(n.line)); + const variants = [variant("full", `// ${rel}\n${text}`, keys)]; + // Symbol span: the lines around the requested definitions, when the atlas knows them. + if (defs.length) { + const from = Math.max(1, Math.min(...defs.map((d) => d.line)) - SPAN_BEFORE); + const to = Math.min(total, Math.max(...defs.map((d) => d.line)) + SPAN_AFTER); + if (from > 1 || to < total) { + const covers = defs.filter((d) => d.line >= from && d.line <= to).map((d) => d.key); + variants.push( + variant( + "span", + `// ${rel}:${from}-${to} of ${total} (definition span)\n${lines.slice(from - 1, to).join("\n")}`, + covers, + keys.filter((k) => !covers.includes(k)), + { shownLines: [from, to], totalLines: total }, + ), + ); + } + } + if (total > HEAD_LINES) { + // The head covers a definition only if its declaration line is inside the head; a whole + // file or test file is never "covered" by its first 25 lines. + const covers = needs + .filter((n) => n.kind === "def" && Number.isFinite(n.line) && n.line <= HEAD_LINES) + .map((n) => n.key); + variants.push( + variant( + "head", + `// ${rel} (first ${HEAD_LINES} of ${total} lines)\n${lines.slice(0, HEAD_LINES).join("\n")}`, + covers, + keys.filter((k) => !covers.includes(k)), + { shownLines: [1, HEAD_LINES], totalLines: total }, + ), + ); + } + variants.push(variant("pointer", `- read ${rel}`, [], keys)); + const ladder = []; + for (const v of variants.sort((a, b) => b.tokens - a.tokens)) + if (!ladder.length || v.tokens < ladder[ladder.length - 1].tokens) ladder.push(v); + return { id: `${source}:${rel}`, source, covers: keys, score, variants: ladder }; } /** * Assemble the context for a task: pinned required items (downgraded before dropped), * optional items greedily by value density, and the missing set as derived questions. + * + * Honesty contract (review F02/F03): + * - `tokens` is measured on the RENDERED block (labels and separators included) with the + * chars/3.6 estimate (`tokenEstimate` says so). The block never exceeds `budget` by that + * measure: when even pointers cannot fit, required items are DROPPED (lowest score first) + * and reported, with `overflow: true`. + * - `covered` holds only keys whose content was actually delivered; `pending` holds keys the + * block merely points at (a pointer, or a partial span/head) — read obligations; `missing` + * holds keys neither delivered nor pointed at (unresolvable, or dropped on overflow). + * - `ok` means every required key was DELIVERED within budget: no missing, no pending, no + * overflow. It is syntactic delivery, not semantic sufficiency. + * - Optional items are chosen greedily by value density (score per token) with per-source + * diminishing returns — a heuristic, with no knapsack or set-cover guarantee (the + * per-source discount breaks the preconditions those guarantees need). * @param {string} root * @param {string} task * @param {{budget?:number, atlas?:any, claims?:any[], nowDay?:number}} [opts] @@ -131,29 +215,46 @@ export function assemble( const required = requiredSet(root, task, { atlas, claims: allClaims, nowDay }); // --- build candidate items, keyed by what they cover ------------------------------- + // File-backed keys are grouped per file first, so one file is one item whose variants + // know exactly which of its keys each one delivers. + /** @type {Map} */ + const files = new Map(); + const need = (rel, n, source, score) => { + const f = files.get(rel) ?? { needs: [], source, score }; + f.needs.push(n); + if (score > f.score) { + f.score = score; + f.source = source; + } + files.set(rel, f); + }; const items = []; + /** @type {{id:string, shown:number, total:number, omitted:string[]}[]} */ + const truncated = []; for (const r of required) { if (!r.resolvable) continue; if (r.kind === "def") { const hit = atlasQuery(atlas, r.name).find((s) => s.name === r.name || s.qname === r.name); - if (hit?.file) { - const it = fileItem(root, hit.file, { covers: [r.key], source: "def", score: 1 }); - if (it) items.push(it); - } + if (hit?.file) need(hit.file, { key: r.key, kind: "def", line: hit.line }, "def", 1); } else if (r.kind === "file") { - const it = fileItem(root, r.name, { covers: [r.key], source: "def", score: 1 }); - if (it) items.push(it); + need(r.name, { key: r.key, kind: "file" }, "def", 1); } else if (r.kind === "tests") { - const it = fileItem(root, r.name, { covers: [r.key], source: "tests", score: 0.9 }); - if (it) items.push(it); + need(r.name, { key: r.key, kind: "tests" }, "tests", 0.9); } else if (r.kind === "deps" && atlas) { - const hop1 = impact(atlas, r.name, { maxHops: 1 }) - .impacted.filter((x) => x.hopDistance === 1) - .slice(0, 12); + const hop1 = impact(atlas, r.name, { maxHops: 1 }).impacted.filter( + (x) => x.hopDistance === 1, + ); + const shown = hop1.slice(0, DEPS_SHOWN); + const omitted = hop1.slice(DEPS_SHOWN).map((x) => `${x.node.name} (${x.node.file})`); + if (omitted.length) + truncated.push({ id: `deps:${r.name}`, shown: shown.length, total: hop1.length, omitted }); const text = hop1.length ? [ `direct dependents of ${r.name} (edit these with it or verify them):`, - ...hop1.map((x) => ` - ${x.node.name} (${x.node.file}, via ${x.edgeKinds[0]})`), + ...shown.map((x) => ` - ${x.node.name} (${x.node.file}, via ${x.edgeKinds[0]})`), + ...(omitted.length + ? [` … and ${omitted.length} more, omitted here (\`forge impact ${r.name}\`)`] + : []), ].join("\n") : `no direct dependents of ${r.name} found in the atlas`; items.push({ @@ -161,22 +262,33 @@ export function assemble( source: "deps", covers: [r.key], score: 1, - variants: [{ gran: "full", text, tokens: tokensOf(text) }], + variants: [ + variant("full", text, [r.key]), + variant("pointer", `- dependents: \`forge impact ${r.name}\``, [], [r.key]), + ], }); } else if (r.kind === "lesson") { const c = allClaims.find((x) => x.id === r.name); if (c) { const text = `lesson (val ${val(c, nowDay).toFixed(2)}): ${c.body.correctedBehavior}`; + const id8 = c.id.slice(0, 8); items.push({ - id: `lesson:${c.id.slice(0, 8)}`, + id: `lesson:${id8}`, source: "lesson", covers: [r.key], score: 0.95, - variants: [{ gran: "full", text, tokens: tokensOf(text) }], + variants: [ + variant("full", text, [r.key]), + variant("pointer", `- lesson: \`forge ledger show ${id8}\``, [], [r.key]), + ], }); } } } + for (const [rel, f] of files) { + const it = fileItem(root, rel, f); + if (it) items.push(it); + } // Optional extras: trusted scope-matching facts (nice-to-have, never required). for (const c of allClaims) { if (c.kind !== "fact" || c.tombstone) continue; @@ -188,37 +300,51 @@ export function assemble( source: "fact", covers: [], score: 0.3 + 0.4 * v, - variants: [{ gran: "full", text, tokens: tokensOf(text) }], + variants: [variant("full", text, [])], }); } - // A symbol's definition and an explicitly named file often resolve to the SAME file — - // merge items by id (union of covered keys) so one span never gets injected twice. - const byId = new Map(); - for (const it of items) { - const prev = byId.get(it.id); - if (prev) { - prev.covers = [...new Set([...prev.covers, ...it.covers])]; - prev.score = Math.max(prev.score, it.score); - } else byId.set(it.id, it); - } - const merged = [...byId.values()]; - // --- selection: pin required coverage, downgrade before dropping ------------------- - const pinned = merged.filter((i) => i.covers.length); - const optional = merged + const pinned = items.filter((i) => i.covers.length).sort((a, b) => (a.id < b.id ? -1 : 1)); + // Value density: score per token of the item's full variant (ties by id, deterministic). + const density = (i) => i.score / Math.max(1, i.variants[0].tokens); + const optional = items .filter((i) => !i.covers.length) - .sort((a, b) => b.score - a.score || (a.id < b.id ? -1 : 1)); - const chosen = pinned.map((i) => ({ item: i, v: 0 })); // v = variant index - const used = () => chosen.reduce((n, c) => n + c.item.variants[c.v].tokens, 0); + .sort((a, b) => density(b) - density(a) || (a.id < b.id ? -1 : 1)); + let chosen = pinned.map((i) => ({ item: i, v: 0 })); // v = variant index + const sepTokens = tokensOf(SEP); + // Per-piece ceilings plus separators bound the rendered block's estimate from above. + const used = () => + chosen.reduce((n, c) => n + c.item.variants[c.v].tokens, 0) + + Math.max(0, chosen.length - 1) * sepTokens; // Downgrade the largest pinned item one rung at a time until the pins fit the budget. while (used() > budget) { const cand = chosen .filter((c) => c.v < c.item.variants.length - 1) - .sort((a, b) => b.item.variants[b.v].tokens - a.item.variants[a.v].tokens)[0]; - if (!cand) break; // everything is already a pointer — required coverage beats budget + .sort( + (a, b) => + b.item.variants[b.v].tokens - a.item.variants[a.v].tokens || + (a.item.id < b.item.id ? -1 : 1), + )[0]; + if (!cand) break; cand.v++; } + // Everything is already at its smallest rung and still does not fit: the budget cannot + // hold the required set. Drop pinned items (lowest score first, then largest) and SAY so — + // never return a block over budget, and never call it complete (F02). + /** @type {string[]} */ + const dropped = []; + while (used() > budget && chosen.length) { + const worst = [...chosen].sort( + (a, b) => + a.item.score - b.item.score || + b.item.variants[b.v].tokens - a.item.variants[a.v].tokens || + (a.item.id < b.item.id ? 1 : -1), + )[0]; + chosen = chosen.filter((c) => c !== worst); + dropped.push(worst.item.id); + } + const overflow = dropped.length > 0; // Greedy fill by value density with per-source diminishing returns. The cut is checked // BEFORE taking an item, and skips only that source — other sources keep competing // (a `break` here used to end the whole fill, and only after taking a 6th item). @@ -226,14 +352,20 @@ export function assemble( for (const item of optional) { const taken = perSource[item.source] ?? 0; if (SOURCE_DISCOUNT ** taken < SOURCE_VALUE_FLOOR) continue; // 4th+ from this source - const variant = item.variants[0]; - if (used() + variant.tokens > budget) continue; + const v = item.variants[0]; + if (used() + v.tokens + (chosen.length ? sepTokens : 0) > budget) continue; chosen.push({ item, v: 0 }); perSource[item.source] = taken + 1; } - const covered = new Set(chosen.flatMap((c) => c.item.covers)); - const missing = required.filter((r) => !r.resolvable || !covered.has(r.key)); + const block = chosen.map((c) => c.item.variants[c.v].text).join(SEP); + const covered = new Set(chosen.flatMap((c) => c.item.variants[c.v].covers)); + const pending = new Set( + chosen.flatMap((c) => c.item.variants[c.v].pending).filter((k) => !covered.has(k)), + ); + const missing = required.filter( + (r) => !r.resolvable || (!covered.has(r.key) && !pending.has(r.key)), + ); const questions = missing .filter((r) => !r.resolvable) .map((r) => @@ -241,32 +373,57 @@ export function assemble( ? `The task names \`${r.name}\` but the repo doesn't define it — which file implements it (or is it new)?` : `The task names \`${r.name}\` but that file doesn't exist — where should this live?`, ); + for (const c of chosen) { + const t = c.item.variants[c.v].truncated; + if (t) truncated.push({ id: c.item.id, ...t }); + } return { - ok: missing.length === 0, + ok: missing.length === 0 && pending.size === 0 && !overflow, budget, - tokens: used(), + tokens: tokensOf(block), + tokenEstimate: TOKEN_ESTIMATE, + overflow, + ...(overflow ? { dropped } : {}), required: required.map((r) => r.key), covered: [...covered].sort(), + pending: [...pending].sort(), missing: missing.map((r) => r.key), questions, + truncated, selection: chosen.map((c) => ({ id: c.item.id, source: c.item.source, gran: c.item.variants[c.v].gran, tokens: c.item.variants[c.v].tokens, + covers: c.item.variants[c.v].covers, })), - block: chosen.map((c) => c.item.variants[c.v].text).join("\n\n"), + block, }; } /** Human rendering for `forge context`. */ export function renderContext(r) { const lines = ["Forge context — budgeted assembly + completeness gate", ""]; + const state = r.ok ? "COMPLETE" : r.overflow ? "OVER BUDGET — INCOMPLETE" : "INCOMPLETE"; lines.push( - ` budget: ${r.tokens}/${r.budget} tokens · required ${r.required.length} · ${r.ok ? "COMPLETE" : "INCOMPLETE"}`, + ` budget: ${r.tokens}/${r.budget} tokens (${r.tokenEstimate ?? TOKEN_ESTIMATE}) · required ${r.required.length} · ${state}`, ); for (const s of r.selection) lines.push(` + ${s.id} [${s.gran}] ${s.tokens}t`); + if (r.overflow) + lines.push( + "", + ` over budget: even pointers could not fit — dropped ${(r.dropped ?? []).join(", ")}`, + ); + if (r.pending?.length) { + lines.push("", " pending reads (pointed at, not delivered — read before acting):"); + for (const p of r.pending) lines.push(` - ${p}`); + } + for (const t of r.truncated ?? []) + if (t.omitted?.length) + lines.push( + ` ~ ${t.id}: ${t.omitted.length} of ${t.total} omitted (${t.omitted.slice(0, 5).join(", ")}${t.omitted.length > 5 ? ", …" : ""})`, + ); if (r.missing.length) { lines.push("", " missing (computed, not a feeling):"); for (const m of r.missing) lines.push(` - ${m}`); diff --git a/src/cortex.js b/src/cortex.js index 6409ced9..46e3845e 100644 --- a/src/cortex.js +++ b/src/cortex.js @@ -4,7 +4,7 @@ // contradiction. Kept fs-thin and deterministic (day + ids passed in) so it's testable // without any hook wiring. -import { recordLessonEvent, supersedeLessonClaim } from "./ledger_bridge.js"; +import { equivalentLesson, recordLessonEvent, supersedeLessonClaim } from "./ledger_bridge.js"; import { ledgerLessons, mergedLessons } from "./ledger_read.js"; import { recordUse, repoLedger } from "./ledger_store.js"; import { @@ -245,15 +245,45 @@ export function applyDistillation(root, lessonId, distilled) { // supersede below rewrite it there (save() is a ledger-only no-op that still succeeds). const lesson = localLessons(root, 0).find((l) => l.id === lessonId); if (!lesson) return false; - const updated = { + const rewritten = { ...lesson, whatWentWrong: distilled.whatWentWrong, correctedBehavior: distilled.correctedBehavior, }; + // A model rewrite is a PROPOSAL (review F07): unless it is equivalent by the narrow rule + // (same text up to case/whitespace/punctuation, no semantic conflict), the lesson's earned + // standing does not carry over to the new wording — it restarts as a candidate that has to + // earn its own confirmations. What it earned before stays inspectable in provenance and on + // the ledger's superseded parent claim. + const hadEvidence = + (lesson.evidenceCount ?? 0) > 0 || + (lesson.contradictionCount ?? 0) > 0 || + lesson.status === "active"; + const updated = + equivalentLesson(lesson, rewritten) || !hadEvidence + ? rewritten + : { + ...rewritten, + status: "candidate", + evidenceCount: 0, + contradictionCount: 0, + quarantineReconfirms: 0, + lastConfirmedDay: lesson.createdDay ?? lesson.lastConfirmedDay, + provenance: { + ...(lesson.provenance ?? {}), + rewrittenFrom: { + whatWentWrong: lesson.whatWentWrong, + correctedBehavior: lesson.correctedBehavior, + status: lesson.status, + evidenceCount: lesson.evidenceCount ?? 0, + contradictionCount: lesson.contradictionCount ?? 0, + }, + }, + }; const ok = save(root, updated).ok; - // A body rewrite changes the content-addressed claim id — supersede in the ledger - // (mint the distilled claim, carry the evidence over, tombstone the template claim) - // or the lesson's history splits across two disjoint claims. + // A body rewrite changes the content-addressed claim id — supersede in the ledger (mint + // the distilled claim, tombstone the template claim as its parent; evidence carries over + // only for an equivalent rewrite) so the lesson's history stays linked, not split. if (ok) supersedeLessonClaim(root, lesson, updated); return ok; } diff --git a/src/cost_report.js b/src/cost_report.js index 6bcb275d..40fe17db 100644 --- a/src/cost_report.js +++ b/src/cost_report.js @@ -351,7 +351,8 @@ export function estimateSpendFromLogs({ root = process.cwd(), fetchImpl, date } else unpriced.push(model); modelBreakdown.push({ model, - cost, + // An unpriced model's cost is UNKNOWN, not zero (review A10). + cost: pricing ? cost : null, priced: Boolean(pricing), priceSource: pricing ? `${pricing.source}${pricing.basis ? `:${pricing.basis}` : ""}` @@ -362,8 +363,20 @@ export function estimateSpendFromLogs({ root = process.cwd(), fetchImpl, date } cacheReadTokens: u.cacheReadTokens, }); } - modelBreakdown.sort((a, b) => b.cost - a.cost); - return { totalCost, sessions, byModel: modelBreakdown, unpriced }; + modelBreakdown.sort((a, b) => (b.cost ?? -1) - (a.cost ?? -1)); + // Label what this number IS (review A10): an estimate — logged tokens × published + // per-token prices, in USD — not an invoice; complete only when every model was priced; + // and it never includes verifier/test runtime or non-model tool costs. + return { + totalCost, + currency: "USD", + basis: "estimate: logged tokens × published per-token prices (not an invoice)", + complete: unpriced.length === 0, + excludes: ["verifier/test runtime", "tool and API costs outside model calls"], + sessions, + byModel: modelBreakdown, + unpriced, + }; } catch { return null; } diff --git a/src/dash.html b/src/dash.html index 356e15d8..abf4fcc6 100644 --- a/src/dash.html +++ b/src/dash.html @@ -3,6 +3,7 @@ + forge dash

    SourceIDGradeNote
    Cognitive Architectures for Language Agents
    Theodore R. Sumers, Shunyu Yao, Karthik Narasi, 2023
    2309.02427confirmedRetrieved via arXiv metadata API; title/authors match claim exactly. Unifies memory, planning/reasoning, action, and learning modules into a single CoALA framework for language agents, giving the cognitive-substrate work's memory/im…