From 2699aa02397ef148b9d6499d014b9fa59494eee9 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 00:42:37 +0000 Subject: [PATCH 1/5] fix: close the 2026-09-27 follow-up review findings (N01-N08) N03 - a root test script covers the workspaces only when it is a recognized, unfiltered recursive run of every member's test script whose failure reaches the exit status; node -r, filters and masked runs cover nothing. Coverage is labelled measured, inferred or declared. N01 - exact reuse is byte-exact (key v3: a digest of the spec's code units). The near tier's semantic guard compares code layout and no longer folds Unicode; inline code the ledger would rewrite is refused at mint. N08 - a dependency contract is its whole declaration minus the body (the atlas records definition extents, v5), bound to the module the artifact imports. N02 - consolidation and ledger compaction merge or archive exact duplicates only; near-duplicates are proposed, and only an exact refuted claim drops a lesson. N06/N07 - verifier events carry a derived authenticated flag (none without a key; the v2 MAC covers every field), and outcome provenance is re-derived on every read from authenticated events with a matching verdict. N04 - a context span delivers a definition only whole; cut ones stay pending and are listed under `partial`. N05 - nested repositories and submodules are bound by their own code state (manifest-v3); unbound code turns a PASS into INCOMPLETE unless declared in verify.external. Also: maxPossibleCost is renamed estimatedCostIfAllAttemptsRun; the claim registry records assessed_release and review counterexamples, and bump.mjs stamps unreleased claims; seeded property tests along the semantic boundaries; linear scans replace two backtracking regexes. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- ARCHITECTURE.md | 8 +- CHANGELOG.md | 101 ++++ README.md | 3 +- docs/GUIDE.md | 103 ++-- docs/UNIVERSAL_ROUTING.md | 4 +- docs/plans/substrate-v2/03-reuse-cache.md | 10 +- .../plans/substrate-v2/04-context-assembly.md | 13 +- docs/status/README.md | 98 ++-- docs/status/claims.json | 73 ++- mintlify/changelog/overview.mdx | 32 ++ mintlify/cli/memory.mdx | 20 +- mintlify/cli/quality.mdx | 16 +- mintlify/cli/substrate.mdx | 4 +- mintlify/concepts/model-routing.mdx | 8 +- mintlify/concepts/proof-carrying-memory.mdx | 7 +- mintlify/concepts/verification-gates.mdx | 23 +- .../cognitive-substrate/EXECUTIVE_SUMMARY.md | 15 +- scripts/bump.mjs | 24 + scripts/claims-status.mjs | 149 +++++- src/atlas.js | 24 +- src/cli/memory.js | 21 +- src/cli/routing.js | 5 +- src/cli/verification.js | 17 + src/context.js | 122 +++-- src/learn_consolidate.js | 133 ++++-- src/ledger.js | 4 + src/ledger_retention.js | 72 +-- src/reuse.js | 360 +++++++++++--- src/router/index.js | 108 ++++- src/router/policy.js | 11 +- src/semantic_guard.js | 117 ++++- src/stack.js | 449 +++++++++++++++++- src/verify.js | 304 +++++++++--- test/bump.test.js | 45 ++ test/claims_status.test.js | 72 +++ test/context.test.js | 89 ++++ test/learn_consolidate.test.js | 117 ++++- test/ledger_retention.test.js | 54 ++- test/reuse.test.js | 238 +++++++++- test/router_math.test.js | 7 +- test/router_universal.test.js | 125 ++++- test/semantic_guard.test.js | 14 + test/stack.test.js | 142 +++++- test/trust_properties.test.js | 251 +++++++++- test/verify.test.js | 181 ++++++- 45 files changed, 3296 insertions(+), 497 deletions(-) diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 58d68930..619a147e 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -148,7 +148,7 @@ The completeness gate on the retrieval side is `forge context ""`: it pins required-knowledge set for the edit (`R(edit)`), downgrades items along a compression ladder before dropping anything, fills the rest of the budget with a value-density heuristic (no approximation guarantee), and reports the _computed missing set_ — the inputs it could not -assemble — plus pending reads and truncations. Its "complete" means syntactically delivered +assemble — plus pending reads, partially shown definitions and truncations. Its "complete" means syntactically delivered within an estimated token budget, not semantically sufficient ([plan 04 §7](docs/plans/substrate-v2/04-context-assembly.md#7-status-2026-09-26--partial)). That missing set is exactly what the substrate pipeline's context stage reads to decide whether an edit is safe to start. Surface: `forge reuse query | mint | stats`. @@ -558,10 +558,10 @@ forgekit/ emit/ # one module per tool (claude, codex, cursor, gemini, aider, copilot, windsurf, zed, continue) + mcp ledger.js # PCM core: content-addressed claims, oracle taxonomy, decayed Beta val, Eq. 3 retrieval, semilattice merge (ADR-0006) ledger_store.js # git-native on-disk ledger (.forge/ledger/): sharded claims, append-only evidence/tombstone logs, normal-form verify, local usage log - ledger_retention.js # retention learned from the ledger's own history: archive never-served claims, idle ones past the longest observed comeback, and BIC-detected near-duplicates (`ledger compact`) + ledger_retention.js # retention learned from the ledger's own history: archive never-served claims, idle ones past the longest observed comeback, and exact duplicates; BIC-detected near-duplicates are only proposed (`ledger compact`) ledger_bridge.js # legacy-store bridge, dormant by default (ledger-only); `FORGE_LEDGER_ONLY=0` re-enables cortex/recall/brain shadow-writes + idempotent `ledger import` ledger_read.js # ledger-only read path by default (`FORGE_LEDGER_ONLY=0` merges legacy∪ledger instead): cortex lesson/fact injection, `recall list`, brain's AGENTS.md index all see teammate knowledge from `ledger merge` - learn_consolidate.js # bin/learn-consolidate.sh: deterministic consolidation of ~/.claude/skills/learned — merge duplicates, drop only ledger-refuted (dormant/retracted/attic) lessons; no model call + learn_consolidate.js # bin/learn-consolidate.sh: deterministic consolidation of ~/.claude/skills/learned — merge exact duplicates, propose near-duplicates, drop only lessons whose exact ledger claim is refuted (dormant/retracted/attic); no model call reuse.js # proof-carrying artifact cache: fingerprint (MinHash+LSH), exact→near→adapt→miss ladder, atlas revalidation embed.js # optional embeddings tier (ADR-0005): FORGE_EMBED=cmd:|http:, swaps MinHash/Jaccard for cosine in `reuse query`/`ledger query`, disk-cached at .forge/embed-cache.jsonl, silent fallback to MinHash context.js # budgeted context assembly + completeness gate: R(edit) coverage, compression ladder, computed missing-set @@ -646,7 +646,7 @@ flowchart LR scripts["scripts
5 files"] docs["docs
1 file"] examples["examples
1 file"] - test -- 286 --> src + test -- 290 --> src bench -- 12 --> src scripts -- 5 --> src examples -- 4 --> src diff --git a/CHANGELOG.md b/CHANGELOG.md index 50942dc1..38bfe192 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,107 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +Fixes for the eight findings of the 2026-09-27 follow-up review (N01–N08) and its suggestions. +The review's reproduction script (`reproduce_remaining.mjs --assert-fixed`) and the original +review's script (`original-reproduce.mjs --assert-fixed`) both pass. + +### Fixed + +- **Node's `-r` preload flag no longer passes as a recursive workspace run (N03).** A root + `test` script covered the workspaces whenever a flag like `-r` or `--workspaces` appeared + anywhere in it, so `node -r ./setup.cjs --test` skipped a failing workspace and `forge verify` + said PASS. The script is now read as shell structure: commands, quotes, wrappers + (`env`, `cross-env`, `npx`, `pnpm exec`) and exit-status flow. Only a recognized, unfiltered + recursive run of every member's `test` script counts: `npm test --workspaces`, `pnpm -r test`, + `yarn workspaces foreach -A run test`, `turbo run test`, `lerna run test` or + `nx run-many -t test`, directly or through `npm run` hops. Filters (`--filter`, `--scope`, + `-w`), masked or backgrounded runs and look-alike flags on other tools cover nothing, and each + package then runs its own suite. Membership is read from the tool's own workspace list, + `!` negations included, and a Python or Go suite inside a workspace is never inferred from an + npm-style run. +- **Exact reuse is byte-exact (N01).** The exact key collapsed whitespace and folded Unicode, + so an artifact minted for `return "a b"` was served as-is for `return "a b"`. The exact tier + now compares a digest of the spec's code units (key version 3), with nothing normalized at + that boundary. Whitespace inside literals, tabs and newlines, escapes, indentation and + composed vs decomposed characters all count. Artifacts keyed before v3 never exact-hit. + Inline code that the ledger's canonical storage would rewrite (CRLF, non-NFC text) is refused + at mint instead of being silently folded. +- **A dependency's whole declaration is its contract (N08).** The contract hash stopped at the + first `{`, so `calc({a})` and `calc({b})` hashed the same and an artifact whose dependency + changed kept serving as valid. The declaration is now read whole (the atlas records each + definition's extent) and lexed, and only the body is cut. Destructured keys, defaults, nested + patterns, return annotations, overload sets and a value's definition all count; comments and + reformatting do not. The dependency is the definition the artifact's own import binds to, not + the first same-name symbol by file order. A contract that cannot be established (an + ambiguous name, an older format) is reported unknown, never valid. +- **Similar rules are never merged or dropped (N02).** Lesson consolidation merged + near-duplicates that passed the semantic guard, but "Allow admins and deny guests…" and "Deny + admins and allow guests…" share every token. Now only exact duplicates merge (the same + statement up to whitespace and trailing punctuation). Near-duplicates are kept and listed as + `proposed` for a person to merge. A lesson is dropped only when a ledger claim with exactly + its text is refuted; one that merely resembles a refuted claim is kept and `flagged`. + `forge ledger compact` follows the same rule: it archives exact duplicates only and lists + near-duplicates for review. +- **An unsigned verifier event never earns verified provenance (N06).** When no evidence key + could be read or created, `readVerifyEvents` accepted any unsigned line, and `forge route + outcome --verify-run` labelled the outcome `verify-event`. Events now carry a derived + `authenticated` flag, false for every event when no key is available. An outcome citing an + unauthenticated event is still recorded, but as self-reported with a `provenanceNote`. +- **An edited provenance label counts for nothing (N07).** `readOutcomes` trusted the stored + `provenance` field. It now re-derives every row's provenance from the authenticated events: + a missing or unauthenticated run, a verdict mismatch, or a second attempt citing the same run + reads as self-reported, and is counted in `readOutcomes.lastDowngraded`. One verifier run + backs one outcome, and the fitter's verified counter uses only derived labels. +- **A long function is delivered whole or not at all (N04).** A context span that showed lines + 1–41 of a 103-line function counted the definition as delivered and reported COMPLETE. A span + or head now covers a definition only when it shows the whole definition. A cut one stays a + pending read and is named under `partial`, with the lines shown and the lines it spans. Small + definitions are still delivered within a small budget. +- **Code in a nested repository is bound by the fingerprint (N05).** An untracked embedded + repository was bound by path only, so a test could rewrite code it imported from there and the + stamp still matched. Nested repositories and submodule working trees are now bound by their + own HEAD, diffs and untracked files, recursively, and the review's mutation fixture is + INCOMPLETE. Code that still cannot be bound is listed as `unbound` in the result and the + verifier event, and it turns a PASS into INCOMPLETE. + +### Added + +- **`verify.external`** in `.forge/forge.config.json`: paths deliberately outside the verified + code, such as a vendored checkout or a nested repository you do not own. They are excluded + from the fingerprint, and every verifier event records them. +- **Coverage basis in `forge verify`.** Each package's verdict is labelled `measured` (its own + suite ran), `inferred` (from a recognized recursive root command, named in `rootRun`) or + `declared` (`verify.workspaces: "root"`), and the CLI prints a `coverage basis` line. +- **Verifier event contract v2.** The event's MAC now covers every field in canonical form: + suites, coverage and its basis, pre/post state, unbound paths, declared external paths and + the environment. Previously it covered only the run id, verdict and code state. Readers see + `authScope` (`event` or, for v1 events, `verdict`). +- **The claim registry is a release artifact.** Every claim now records `assessed_release`, + the release that ships its assessed commit, and can link review `counterexamples`. + `scripts/bump.mjs` stamps claims marked `unreleased` with the version it cuts. With release + tags present, `node scripts/claims-status.mjs --check` fails on a stale `unreleased` or on a + release that does not contain the assessed commit. Twelve entries still read "master after + 1.4.3 (unreleased)"; they now name 1.5.0, the release that shipped them. +- **Property tests along semantic boundaries.** Seeded families for quoted whitespace and code + points, subject and number swaps, look-alike and filtered workspace commands, destructured + parameters, and a missing evidence key. + +### Changed + +- **`maxPossibleCost` is now `estimatedCostIfAllAttemptsRun`.** It sums each attempt's expected + cost, so it was never a bound on what a run can bill. The CLI says "an estimated $X if every + attempt runs". Consumers of the `forge route universal --json` field need the new name. +- **Formats that changed rebuild or re-verify on their own.** The atlas (version 5) records + definition extents, the code-state fingerprint is `manifest-v3`, reuse keys are version 3 and + dependency contracts are `v2:`. A stamp from the older fingerprint no longer verifies, so + re-run `forge verify`. +- **The semantic guard compares code layout and stops folding Unicode.** Fenced blocks, + indented code lines and unusual spacing beside a code token are compared verbatim, and + literals, identifiers and paths are compared by code point. +- **The research executive summary marks its 2026-09-26 correction in place.** The "cannot be + prompted or tooled away" thesis is labelled too broad, and the whitepaper PDF is labelled as + the pre-correction edition. + ## [1.7.2] - 2026-09-26 ### Changed diff --git a/README.md b/README.md index 313e52a2..847041ba 100644 --- a/README.md +++ b/README.md @@ -218,7 +218,8 @@ from a fresh repository graph. - **Budgeted context assembly.** Definitions, direct dependants, sibling tests, and trusted lessons are selected under a token budget. Missing required context becomes a question rather than invented context. Coverage is syntactic — delivered, not proven sufficient — token - counts are estimates, and a file that only fits as a "read this" pointer stays a pending read. + counts are estimates, a file that only fits as a "read this" pointer stays a pending read, + and a definition counts as delivered only when its whole body is. - **Model-tier recommendation.** A deterministic rubric combines task text and repository signals. An optional LLM proposal can only lower the tier, confidence-gated and bounded; a vote for a higher tier is never applied automatically — it is recorded as an advisory diff --git a/docs/GUIDE.md b/docs/GUIDE.md index d9bd6e33..0638cc11 100644 --- a/docs/GUIDE.md +++ b/docs/GUIDE.md @@ -233,8 +233,9 @@ recommendation is labelled `advice only` (`applicable: false`, `unmapped` in `-- Being in the registry does not make a model callable. Add its id under `providers` in `.forge/models.json`, or pass `--provider ` to route only among models you can call. `forge route outcome` records one attempt's pass/fail, labelled self-reported unless -`--verify-run ` ties it to a `forge verify` run whose verdict agrees; `--attempt ` -makes recording idempotent. The model, its evidence status and its limits: +`--verify-run ` ties it to an authenticated `forge verify` run whose verdict agrees; +`--attempt ` makes recording idempotent. The label is re-derived from the verifier events +every time outcomes are read, so editing it in the file changes nothing. The model, its evidence status and its limits: [docs/UNIVERSAL_ROUTING.md](UNIVERSAL_ROUTING.md). ### `forge models` — what each tier resolves to @@ -692,35 +693,58 @@ Forge verify own directory: the root, plus each nested package with an explicit `scripts.test`, a pytest config, a `go.mod`, and so on. `packages: n/m covered` counts the packages that reached a verdict. If any package's suite never reaches one, the result is `INCOMPLETE`, never `PASS`. -If any package fails, the result is `FAIL`, however green the root is. A root script that -already runs every workspace (`npm test --workspaces`, `pnpm -r test`, `yarn workspaces -foreach`, `turbo run test`, `lerna run test`, `nx run-many`) covers them in one run instead -of once per package. Fixture and test-data packages are never required. A test runner that is +If any package fails, the result is `FAIL`, however green the root is. A root `test` script +covers the workspaces in one run only when forge can establish that it runs every member's +own `test` script. It reads the script as shell commands and recognizes an unfiltered +recursive run whose failure reaches the script's exit status: `npm test --workspaces`, +`pnpm -r test`, `yarn workspaces foreach -A run test`, `turbo run test`, `lerna run test` or +`nx run-many -t test`, directly or through `npm run` hops. A filtered run (`--filter`, +`--scope`, `-w`), a masked or backgrounded one, and a look-alike flag such as Node's preload +`node -r` cover nothing: each package then runs its own suite. Only members of that tool's +own workspace list count, `!` negations included. The `coverage basis` line says what each +package's verdict rests on: `measured` (its own suite ran), `inferred` (the recognized root +command) or `declared` (`workspaces: "root"`). Fixture and test-data packages are never +required. A test runner that is only a devDependency is not an obligation either; `forge stack` lists it as `available`. Tune this per repo under `verify` in `.forge/forge.config.json`: ```json -{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"] } } +{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"], "external": ["vendor/upstream"] } } ``` -- `workspaces: "root"` declares that the root command already covers every package. +- `workspaces: "root"` declares that the root command already covers every package; those + verdicts are labelled `declared`, never `measured`. - `exclude` lists package paths that are not required suites. - `generated` lists outputs a test run may legitimately write. +- `external` lists paths deliberately outside the verified code, such as a vendored checkout + or a nested repository you do not own. They are left out of the fingerprint, and every + verifier event records them. **Bound to the code that was tested.** The stamp is bound to a fingerprint of the working tree: HEAD, the staged and unstaged diffs, and each untracked file's path, mode, size and content hash. The fingerprint is taken before AND after the run. If something changed the code while the tests ran (a formatter, a code generator, another agent), the result is `INCOMPLETE` -with `mutated: true`, and the stamp names the pre-run state. Interpreter caches -(`__pycache__`, `.pytest_cache`, …) and the `generated` paths never count as a change. An -untracked file that cannot be read makes the state unbindable; it is never silently skipped. -Stamps written before this fingerprint (scheme `manifest-v2`) no longer verify: re-run +with `mutated: true`, and the stamp names the pre-run state. A nested repository (an untracked +embedded repo, or a submodule's working tree) is bound by its own HEAD, diffs and untracked +files, recursively, so a test that rewrites code it imports from one is caught too. +Interpreter caches (`__pycache__`, `.pytest_cache`, …) and the `generated` paths never count +as a change. An untracked file that cannot be read makes the state unbindable; it is never +silently skipped. Code the fingerprint still cannot bind (repositories nested more than three +deep, a socket) is listed as `unbound`, and the result is then `INCOMPLETE`, never `PASS`, +unless the repo declares that path outside what it verifies with `verify.external`. That +declaration excludes it from the fingerprint and is recorded in every verifier event. Stamps +written before this fingerprint (scheme `manifest-v3`) no longer verify: re-run `forge verify`. Every run also appends one event to `.forge/verify-events.jsonl`: the run id, the verifier -and its version, the suites, the coverage, the pre- and post-run state, and an environment -digest, sealed with a machine-local MAC. The event is never rewritten. `forge route outcome ---verify-run ` ties a routing outcome to it. +and its version, the suites, the coverage and its basis, the pre- and post-run state +(unbound paths included), and an environment digest. The event is never rewritten. Its +machine-local MAC (event contract v2) covers every one of those fields in canonical form, so +no field a reader relies on can be edited without the event reading back as unauthenticated. +Readers mark each event `authenticated` or not; without an evidence key (an unwritable state +directory) nothing is authenticated, and an unsigned event never counts as verified evidence. +Events from contract v1 authenticated only the run id, verdict and code state. `forge route +outcome --verify-run ` ties a routing outcome to an authenticated event. **`forge verify --deep` — multi-lens consensus.** The deep mode runs a table of independent lenses over the same diff — the test suite, unknown symbols, atlas @@ -994,12 +1018,15 @@ Forge ledger — compact (every cut-off learned from this ledger) [dry run] claims: 11 · claims with logged use: 10 retention: idle cut-off 4 d = the longest idle stretch any claim came back from (199 comebacks, typical gap 4 d; usage log spans 90 d) - duplicates: boundary 0.28 (two components beat one: BIC -72.6 < 3.5) · 1 group(s) + duplicates: 1 exact group(s) · near-duplicate boundary 0.28 (two components beat one: BIC -72.6 < 3.5) archive: 3 34a49b8d036e idle 86 d > learned cut-off 4 d a5e218fd1814 tombstoned (never served) - d3a5a1c9941e near-duplicate of c70ee7d4f505 (similarity 0.55 ≥ learned 0.28) + d3a5a1c9941e duplicate of c70ee7d4f505 (the same statement) + + kept both — near-duplicates (retract one if they say the same thing): 1 + 9b1e07a2c4d3 ↔ c70ee7d4f505 similarity 0.55 dry run: nothing written ``` @@ -1007,22 +1034,25 @@ Forge ledger — compact (every cut-off learned from this ledger) [dry run] **The three archive rules:** - **Never served:** a tombstoned or dormant claim goes at once, because retrieval never serves it. - **Idle too long:** a live claim goes once it has been idle longer than any claim here has ever been idle and then used again. Until the usage log covers that long, no live claim is archived. -- **Near-duplicates:** each claim's similarity to its closest claim of the same kind is modelled as one group or two, and BIC decides which fits. Only two groups produce a duplicate boundary. The claim kept from each group is the one with the highest val. +- **Duplicates:** claims of one kind that make the same statement (equal up to whitespace and trailing punctuation) keep one survivor, the one with the highest val; the others go. +- **Near-duplicates are never archived.** Each claim's similarity to its closest claim of the same kind is modelled as one group or two, and BIC decides which fits. When two groups fit, pairs above the boundary are listed under `kept both — near-duplicates` for a person to merge. Similarity cannot see which subject gets which action ("allow admins and deny guests" vs "deny admins and allow guests"), a swapped number, or the detail one claim adds, so it only proposes. **Where use comes from:** forge writes `.forge/ledger/.usage.jsonl`, a gitignored local log. It records each claim that the session lesson block, pre-edit lessons, the déjà-vu advisory, `ledger query` or the MCP query served. -**Similar is not the same.** Two claims are grouped as duplicates only when they also agree -on everything that changes behaviour: operators, numbers and units, quoted literals, -identifiers, paths, and negation. "Enable authentication…" and "Disable authentication…" are -never merged, however similar their words. Such a pair is printed under `kept apart — similar -but conflicting`, so a human can retract the wrong one. +**Similar is not the same.** A near-duplicate pair that also differs in something that +changes behaviour (operators, numbers and units, quoted literals, identifiers, paths, +negation, or code layout) is printed under `kept apart — similar but conflicting` instead, +so a human can retract the wrong one: "Enable authentication…" and "Disable +authentication…" are never merged, however similar their words. **What happens to archived claims:** - They move to `.forge/ledger/attic/`, and their logs stay where they are. - Each one records WHY it was archived in `attic/.log`: `tombstoned`, `dormant`, `idle`, or `duplicate` (naming the claim that was kept). Archived is not refuted: consolidation - drops a learned lesson only when its claim was actually retracted or went dormant. An idle - archive keeps the lesson, and a duplicate defers to the claim that was kept. + drops a learned lesson only when a claim with exactly its text was actually retracted or + went dormant. A lesson that is only similar to a refuted claim is kept and flagged, since + it may be the opposite rule. An idle archive keeps the lesson, and a duplicate defers to + the claim that was kept. - `forge ledger show` and `blame` still read them. - New evidence brings one back. - The Stop hook applies the first two rules on its own; duplicates are grouped only by this command. @@ -1105,13 +1135,17 @@ near → adapt → miss. An artifact serves **only while its proof holds**: conf 0.6 floor, its file unchanged since it was minted, and every declared dependency still in the atlas with the same declaration. -- **Exact means the same text.** The key ignores only whitespace (and Unicode - normalization). Case, operators, literals and punctuation all count, so `age >= 18` and - `age <= 18`, or `"ADMIN"` and `"admin"`, never share a key. Artifacts minted before this - key was introduced never hit exact. +- **Exact means the same text, byte for byte.** The key is a digest of the spec as given: + nothing is normalized, because whitespace inside `"a b"`, a Python block's indentation, a + regex's spaces and composed vs decomposed Unicode are all data. Case, operators, literals + and punctuation count too, so `age >= 18` and `age <= 18`, or `"ADMIN"` and `"admin"`, + never share a key. Artifacts minted before key version 3 never hit exact. Inline code the + ledger's storage would rewrite (CRLF line endings, non-NFC text) is refused at mint; mint + it from a file instead. - **Near must also agree on behaviour.** A reworded match is offered as `near` only when - the two specs share their operators, numbers, literals, identifiers, paths and negation. - Otherwise it drops to `adapt`, with a note naming what differs. + the two specs share their operators, numbers, literals, identifiers, paths, negation and + code layout (fenced blocks, indented code lines, and unusual spacing beside a code token + are compared verbatim). Otherwise it drops to `adapt`, with a note naming what differs. - **Checked where it is served.** An artifact whose file was edited or deleted since it was minted is not served. When a dependency's declaration changed, it is not served either. With no atlas to check against, the hit is marked `NOT revalidated` (`requiresRevalidation: @@ -1197,8 +1231,11 @@ What `COMPLETE` (`ok: true`) does and does not mean: separators included — not a model tokenizer's count. - **A pointer is not coverage.** An item that only fits as a one-line `- read ` pointer is a pending read obligation, listed under `pending`, and the assembly is not complete until it - is read. A 25-line head covers a definition only when the definition's line is inside it, and - a dependents list cut at 12 names what it left out (`truncated`). + is read. A definition is delivered only WHOLE, from its declaration to its last line (the + atlas records each definition's extent). A span or 25-line head that shows the declaration + but cuts the body leaves the definition pending and names it under `partial`, with the lines + shown and the lines it spans, so a 104-line function delivered as its first 41 lines is never + reported as delivered. A dependents list cut at 12 names what it left out (`truncated`). - **Over budget is INCOMPLETE.** When even pointers do not fit, the result says `overflow: true` and `ok: false`; it never reports a silent over-budget pass. - **Selection is a heuristic.** Optional items are picked greedily by value density (score ÷ diff --git a/docs/UNIVERSAL_ROUTING.md b/docs/UNIVERSAL_ROUTING.md index cb5af225..cb270cb1 100644 --- a/docs/UNIVERSAL_ROUTING.md +++ b/docs/UNIVERSAL_ROUTING.md @@ -66,10 +66,10 @@ Every ordered cascade of up to 3 candidates is evaluated; `--depth` bounds the s | `value:V` | the cascade that maximises V·P(success) − E[cost] | — | | `budget:B` | the most likely cascade with **expected** cost E[cost] ≤ B | not a runtime spend cap: one task's cascade can cost up to the sum of every attempt in it, and E[cost] itself is under-predicted (see [Modeling limits](#modeling-limits)) | -**When no cascade satisfies the objective.** A `budget:B` or `target:p` that nothing can meet is reported as infeasible, never as a quiet best effort: the result carries `feasible: false`, `budgetMet: false` for a budget objective, the `minimumExpectedCost` any candidate cascade achieves, the `maxPossibleCost` of the recommended cascade (the sum of all its attempts' costs — the worst case), and a `reason`, and `forge route universal` prints an explicit `INFEASIBLE` line. The caller must choose a fallback. Expected cost, the maximum possible cascade cost and the cost actually charged are three different numbers; forgekit advises, and enforcing a hard spend cap belongs to whatever runs the attempts. +**When no cascade satisfies the objective.** A `budget:B` or `target:p` that nothing can meet is reported as infeasible, never as a quiet best effort: the result carries `feasible: false`, `budgetMet: false` for a budget objective, the `minimumExpectedCost` any candidate cascade achieves, the `estimatedCostIfAllAttemptsRun` of the recommended cascade (the sum of every attempt's expected cost — a modeled figure for the path where each stage runs, not a bound on what a run can bill; formerly `maxPossibleCost`), and a `reason`, and `forge route universal` prints an explicit `INFEASIBLE` line. The caller must choose a fallback. Expected cost, the estimated cost if every attempt runs, and the cost actually charged are three different numbers; forgekit advises, and enforcing a hard spend cap belongs to whatever runs the attempts. **4. Learning locally.** -- `forge route outcome "" --model --pass|--fail [--cost ] [--attempt ] [--verify-run ]` records the result of one attempt. Only a hash of the task and its features are stored, never the text. The pass/fail is entered by the caller, so a record is labelled `provenance: "self-reported"`; with `--verify-run`, naming a `forge verify` run in this checkout's verifier-event log whose PASS or FAIL agrees with it, the record is labelled `provenance: "verify-event"` instead (a disagreeing or missing run is refused). Every record carries an `attemptId` (`--attempt`, or a fresh id when omitted), and recording the same attempt id again counts once, so a retry or a replayed outcomes file cannot count one attempt twice. Records are schema-validated when written and when read. +- `forge route outcome "" --model --pass|--fail [--cost ] [--attempt ] [--verify-run ]` records the result of one attempt. Only a hash of the task and its features are stored, never the text. The pass/fail is entered by the caller, so a record is labelled `provenance: "self-reported"`; with `--verify-run`, naming a `forge verify` run in this checkout's verifier-event log whose PASS or FAIL agrees with it, the record is labelled `provenance: "verify-event"` instead (a disagreeing or missing run is refused, and so is a run that already backs another attempt). The event must be authenticated: its MAC must verify under this machine's evidence key. When no key is available, or the MAC does not verify, the outcome is still recorded, but as self-reported with a `provenanceNote`. The stored label is informational only. `readOutcomes` re-derives every row's provenance from the authenticated events on each read, so a hand-edited label, a missing run, a verdict mismatch or a second attempt citing the same run reads as self-reported and is counted in `readOutcomes.lastDowngraded`. A verifier event vouches for the pass/fail verdict only, never for which model produced the patch or what it cost (`verifiedFields: ["passed"]` in the fitted model's provenance). Every record carries an `attemptId` (`--attempt`, or a fresh id when omitted), and recording the same attempt id again counts once, so a retry or a replayed outcomes file cannot count one attempt twice. Records are schema-validated when written and when read. - `forge route fit` refits with the shipped fit as the prior mean. It is a MAP, empirical-Bayes-style **shrinkage** update — a few local outcomes barely move it and many outcomes dominate — not a maintained posterior: it returns a point estimate and carries no covariance over from the shipped fit. - Cost intercepts are updated with a unit-information prior. - A model in the registry but not in the fit enters "cold", at the population-mean ability and loadings, until outcomes arrive. Its point estimate hides a large epistemic uncertainty. diff --git a/docs/plans/substrate-v2/03-reuse-cache.md b/docs/plans/substrate-v2/03-reuse-cache.md index 8a58a3b3..38f1f6eb 100644 --- a/docs/plans/substrate-v2/03-reuse-cache.md +++ b/docs/plans/substrate-v2/03-reuse-cache.md @@ -25,7 +25,8 @@ served — the cache prunes itself by ground truth. ``` artifact.body := { - key: IDENTITY-normalized task text (case/whitespace/punctuation only), + key: the task text as given (IDENTITY text; see §2), + keyHash: sha256 of the task text's code units — the EXACT identity (key version 3), spec: SHAPE-normalized task specification text, sketch: MinHash sketch of spec (for near-match), slice: sha256 of the atlas graph slice the artifact touches, // context key @@ -41,8 +42,11 @@ artifact.body := { **Normalization comes in two forms**, because "the same task" and "the same neighbourhood" are different questions: -- **identity** (`key`): Unicode-aware tokens, lowercased, edge punctuation dropped, - whitespace collapsed — and NOTHING else. Identifiers, paths and numbers stay verbatim. +- **identity** (`key`, `keyHash`): the task text exactly as given. An earlier identity form + lowercased and trimmed punctuation (review F04), and its successor still collapsed + whitespace and folded NFC, so `return "a b"` exact-hit `return "a b"` (review N01). The + exact tier now compares `keyHash`, a digest of the text's code units, and nothing is + normalized at that boundary; similarity search stays layout-blind. - **shape** (`spec`): identity plus typed placeholders for identifiers, paths, numbers and string literals (`⟨ident⟩`, `⟨path⟩`, `⟨num⟩`, `⟨str⟩`). diff --git a/docs/plans/substrate-v2/04-context-assembly.md b/docs/plans/substrate-v2/04-context-assembly.md index 6a8f7e3f..c1b47416 100644 --- a/docs/plans/substrate-v2/04-context-assembly.md +++ b/docs/plans/substrate-v2/04-context-assembly.md @@ -160,9 +160,14 @@ counted as delivering a definition that sat below line 100 of that file (F03). P - **A pointer is an obligation, not coverage.** A one-line `- read ` creates a *pending read obligation*, listed in `pending`; it does not count as covering anything until the span is actually read. -- **A head covers only what it shows.** The "first 25 lines" variant covers a definition only when - the definition's line falls inside the delivered span; definitions use a symbol-span variant when - the atlas knows the line. +- **A definition is delivered whole, or not at all.** A span or the "first 25 lines" head covers a + definition only when it shows the whole definition, from its declaration to its last line (the + atlas records `endLine`, the extent of every definition it can scope). One that shows the + declaration but cuts the body is listed in `partial` (lines shown, lines spanned) and stays a + pending read (review N04: a 103-line function delivered as lines 1–41 used to count as covered). + A definition with no known extent is delivered only by the whole file. The span ladder offers + whole definitions first, so a small definition deep in a large file is still delivered within + a small budget. - **Truncation is visible.** A dependents list cut at 12 entries names the omitted entries in `truncated`. - **Selection is a heuristic.** Optional items are chosen greedily by value density (score ÷ @@ -171,7 +176,7 @@ counted as delivering a definition that sat below line 100 of that file (F03). P - **The ambient hook does not assemble context.** For latency, the per-prompt hook uses caches only; the explicit gate (`forge substrate`) and `forge context` assemble. - **`ok` means syntactically delivered.** Every required key has delivered content within the - budget, with no overflow and no pending reads. It does not mean the delivered text is + budget (every definition whole), with no overflow and no pending reads. It does not mean the delivered text is semantically sufficient for the edit, and `contracts(S)` (the types and interfaces a symbol implements) is not yet part of `R(edit)`. diff --git a/docs/status/README.md b/docs/status/README.md index 23fb96ce..035fbbe1 100644 --- a/docs/status/README.md +++ b/docs/status/README.md @@ -37,53 +37,53 @@ place where every assertion's evidence can be looked up. 47 claims — implemented 13 · measured 10 · partial 5 · reported 3 · hypothesis 6 · refuted 10. Assessed against commits `d2abfa69fb77`, `7eef61179d16` (as of 2026-09-26). -| ID | Status | Claim | Component · version | Evidence | Notes | -| --- | --- | --- | --- | --- | --- | -| `impact-oracle-perfect-recall` | **refuted** | The Python impact oracle has perfect recall: it never misses an affected module. | Python impact oracle (research/python-prototypes/impact_oracle) · v1, as shipped | [`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/impact_oracle/README.md`](../../research/python-prototypes/impact_oracle/README.md) | Recall 1.00 held on five mutations of a ten-file demo package the authors wrote (precision 0.633, F1 0.753). On 759 evaluated files in nine open-source repositories its recall was 0.022. | -| `impact-oracle-field-original` | **measured** | On nine repositories (co-edited-file prediction) the as-shipped oracle scores pooled P/R/F1 0.3982/0.0220/0.0416 against grep's 0.3535/0.5732/0.4373. | Python impact oracle (research/python-prototypes/impact_oracle) · v1, as shipped | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Recomputed by the 2026-09-26 external review: 801 labelled / 759 evaluated files, 20,144 mirrored pairs; repository-cluster bootstrap (20,000 draws, seed 1234) F1 oracle [0.0010, 0.0927], grep [0.3807, 0.5394], grep minus oracle [0.3422, 0.5174]; macro F1 0.0220 vs 0.4947. Co-change is a proxy for historically related edits; regression-test selection was not measured. | -| `impact-oracle-repaired-heldout` | **measured** | The repaired oracle's F1 on three held-out repositories is 0.416 against grep's 0.371 at the canonical threshold 0.02. | Python impact oracle (research/python-prototypes/impact_oracle) · v2, repaired (in-tree, 49 tests) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/impact_oracle/README.md`](../../research/python-prototypes/impact_oracle/README.md) | A point estimate (ΔF1 about +0.044). Per-repository counts exist only at threshold 0.10 (pooled ΔF1 +0.0565, 3/3 repositories favour it, one-sided sign test p = 0.125); pytest supplies 71.3% of held-out pairs. Never mix the 0.02 and 0.10 figures. | -| `impact-oracle-repaired-beats-grep` | **hypothesis** | The repaired oracle beats a grep baseline on repositories it has not seen. | Python impact oracle (research/python-prototypes/impact_oracle) · v2, repaired | [`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/README.md`](../../research/README.md) | Not established: three held-out repositories, and the choice of which relations to add was made after diagnosing all nine (an architecture-selection channel into the test set). Next study: freeze parser and relation design before acquiring new repositories or a time split; report review-cost metrics beside F1. | -| `impact-oracle-in-tree-is-v2` | **implemented** | The in-tree Python impact oracle is the repaired v2, with 49 tests (36 demo-package + 13 repair regressions). | Python impact oracle (research/python-prototypes/impact_oracle) · v2 | [`research/python-prototypes/impact_oracle/tests/test_repair_fixes.py`](../../research/python-prototypes/impact_oracle/tests/test_repair_fixes.py)
[`research/python-prototypes/impact_oracle/tests/test_demo_package.py`](../../research/python-prototypes/impact_oracle/tests/test_demo_package.py) | research/README.md said until 2026-09-26 that the repaired oracle shipped only in the replication archive. ImpactOracle(wm, sibling_enabled=False, forward_enabled=False) reproduces the refuted v1 traversal. | -| `node-impact-fixture-quality` | **measured** | forge impact scores precision 0.17, recall 1.00, F1 0.28 on six hand-labelled cases from this repository. | Node code graph (src/atlas.js, forge impact) · master after 1.4.3 (unreleased) | [`reports/benchmarks.md`](../../reports/benchmarks.md)
[`bench/impact_cases.mjs`](../../bench/impact_cases.mjs) | Six self-labelled symbols in one repository; the transitive walk is scored against direct-only labels. Re-measured on the tree merged as 7eef611, after the CLI handler move added src/cli/memory.js as an eleventh claimText dependent (reports/benchmarks.md: mean F1 0.28, was 0.29). The 2026-09-26 review measured macro 0.18 / 1.00 / 0.30 on the same cases at d2abfa6. A regex-derived graph, not the evaluated Python AST oracle; none of the Python study's numbers apply to it. | -| `router-gate-demo-saving` | **refuted** | Complexity routing saves 62.1% of cost against always using the premium tier. | Old tiered router (research/python-prototypes/router_gate) · July 2026 prototype, thresholds tuned on 30 tasks | [`research/python-prototypes/router_gate/eval_results.json`](../../research/python-prototypes/router_gate/eval_results.json)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Measured on the 30 tasks its thresholds were tuned on, by repricing measured tokens (a price counterfactual). On 80 held-out tasks its total spend was 20.2% higher. | -| `router-gate-heldout-cost` | **measured** | On 64 non-halted held-out tasks the routed pipeline spent $6.3582 against always-premium's $5.2893: 20.21% more. | Old tiered router (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Escalation-inclusive and ungated. 58 of 64 tasks failed at every tier, which makes the inversion largely mechanical. | -| `router-gate-judge-accepted-cost` | **measured** | Per judge-accepted output the routed pipeline cost $1.060 against always-premium's $1.763. | Old tiered router (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/router_gate/README.md`](../../research/python-prototypes/router_gate/README.md) | judge_accepted only: 6/64 vs 3/64 (Clopper-Pearson [0.035, 0.193] and [0.010, 0.131]), so the ratio is unstable. The judge is also the mid-tier executor. tests_passed, human_accepted and deployed_without_revert were not measured. | -| `router-gate-gate-f1` | **measured** | The assumption gate scores F1 0.37 (recall 0.31) on 80 held-out tasks. | Old tiered router's assumption gate (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/empirical-refutation/extended_preprint.html`](../../research/empirical-refutation/extended_preprint.html) | Against 1.00 on the 30-task tuning set. Labels are one model's. | -| `router-gate-label-agreement` | **measured** | Re-labelling 30 held-out tasks gives halt κ 0.5161 and tier κ 0.8919. | Old tiered router evaluation labels · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Same model, reworded prompt: self-consistency, not agreement with a human. | -| `forge-route-tier-cost` | **hypothesis** | The tiered forge route recommendation lowers cost at equal success. | Tiered router (src/route.js, forge route) · 1.4.3 | [`src/route.js`](../../src/route.js)
[`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md) | Never evaluated end to end on cost. A separately versioned descendant of the refuted router_gate rubric; no measured routing factor supports a positive saving. | -| `universal-router-heldout-headline` | **reported** | On 350 held-out SWE-bench Verified issues the universal router (match-best-single) solves 76.3% at $0.093 per task, against 75.1% at $0.364 for the best single model chosen on dev. | Universal router (src/router, forge route universal) · fit on 150 dev issues (harness-bench run 4), not the shipped prior | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Produced by an external harness (harness-bench) that is not shipped here. Its 150/350 split ids, pre-registration, baseline selection and metric aggregation are not in this repository, so it cannot be reproduced from it. The 2026-09-22 re-run (217 of 218 metric values identical) is the project's own report with the same harness, not an independent replication. bench/universal-router/holdout_eval.mjs is a different experiment and does not stand in for it. | -| `universal-router-vs-fixed-cascade` | **measured** | With a 150-task dev fit the universal router does not beat a fixed cascade chosen on the same dev data. | Universal router (src/router, forge route universal) · dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5 | [`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | A new in-repo experiment, not a reproduction of run 4: seeded split 20260926, 150 dev / 350 held-out, fit on dev only, 10,000 paired bootstrap draws. Router 80.0% solved at $0.124 per task; best fixed cascade on dev (minimax-m2.5 > gpt-5-mini > kimi-k2.5) 80.6% at $0.136; router minus cascade -0.6 points [-1.4, 0.0] and -$0.012 [-$0.024, -$0.003]. Five more seeds, run after the first result: -0.6 to +1.1 points, never both more accurate and cheaper. The run-4 version of this finding is repository-reported. | -| `universal-router-inrepo-replay-vs-best-single` | **measured** | On a new in-repo split (seed 20260926; 150 dev / 350 held-out tasks), the router solves 80.0% of held-out tasks at $0.124 per task, against 76.0% at $0.768 for the best single model chosen on dev (claude-opus-4.5). | Universal router (src/router, forge route universal) · dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5 | [`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | +4.0 points [+0.6, +7.4] at -$0.644 per task [-$0.698, -$0.594], mostly from one cheap model: minimax-m2.5 opens 308 of the 350 cascades. It depends on which model wins dev: across six splits the solve-rate interval excludes zero only for this seed. The replay assumes a perfect, free check between attempts; one scaffold, twelve Python repositories, February 2026 costs, and SWE-bench Verified's label-validity limits. | -| `universal-router-prior-refit` | **measured** | The shipped prior (data/router_prior.json) refits exactly from pinned public data with this repository's own tooling. | Universal router prior (data/router_prior.json) · fitted 2026-09-22, k=1, scale 4 | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/reproduce.sh`](../../bench/universal-router/reproduce.sh)
[`bench/universal-router/sources.json`](../../bench/universal-router/sources.json)
[`bench/universal-router/compare_priors.mjs`](../../bench/universal-router/compare_priors.mjs)
[`bench/universal-router/fit_prior.mjs`](../../bench/universal-router/fit_prior.mjs) | Reproduced on 2026-09-26 by the project in its own environment (Node v22.22.2, Python 3.11.15, 4-vCPU Intel Xeon @ 2.80GHz): from an empty work directory all 176 fitted values are identical, only provenance.fittedAt differs, with the router code of d2abfa6 and again with aedddf5; the refit takes 413-424 s. The external review's refit stopped at a 180 s limit, which was too short; the 37.6 s / 45.9 s in the 2026-09-22 report were the 150-issue dev fit. Tasks: Hugging Face SWE-bench/SWE-bench_Verified at revision 78f471b (not princeton-nlp/SWE-bench_Verified). reproduce.sh and sources.json were added after the assessed commit. A calculation reproduction, not an independent third-party replication. | -| `universal-router-prior-transfer` | **hypothesis** | The shipped prior predicts success and cost on a user's own workload. | Universal router prior (data/router_prior.json) · fitted 2026-09-22 | [`data/router_prior.json`](../../data/router_prior.json)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[openai.com/index/why-we-no-longer-evaluate-swe-bench-verified](https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/) | Fitted on all 500 SWE-bench Verified tasks (so it cannot be evaluated out of sample on them), one scaffold, Python repositories, February 2026 costs. OpenAI reports test flaws in 59.4% of an audited 138-problem hard subset (not of all 500) plus contamination evidence. Next: time-held-out or unseen repositories, another scaffold, another language, real spend. | -| `universal-router-target-guarantee` | **refuted** | The target:p objective guarantees a success rate of at least p. | Universal router (src/router, forge route universal) · master after 1.4.3 (unreleased) | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist. Unchanged at 7eef611. | -| `universal-router-budget-contract` | **partial** | The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists. | Universal router (src/router, forge route universal) · master after 1.4.3 (unreleased) | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`src/router/policy.js`](../../src/router/policy.js) | B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). The explicit half now holds at 7eef611: an infeasible budget returns ok:false, feasible:false, budgetMet:false, minimumExpectedCost and a reason (the CLI prints INFEASIBLE), with the least-bad cascade only as a labeled fallback, and every recommendation reports maxPossibleCost (review F12, regression-tested). | -| `universal-router-cascade-cost-independence` | **refuted** | A later cascade attempt costs the same in expectation whether or not earlier attempts failed. | Universal router cost model (src/router/policy.js) · master after 1.4.3 (unreleased) | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | The cascade cost formula assumes E[cost_i \| earlier attempts failed, x] = E[cost_i \| x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes. Unchanged at 7eef611. | -| `route-outcome-provenance` | **partial** | Outcomes recorded with forge route outcome are verified results. | Universal router outcome log (src/router/index.js) · master after 1.4.3 (unreleased) | [`src/router/index.js`](../../src/router/index.js)
[`test/router_universal.test.js`](../../test/router_universal.test.js)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | Re-assessed on 7eef611: records are schema-validated on write and read, recording is idempotent by attemptId (--attempt), and a row is provenance verify-event only when --verify-run names a forge verify run whose verdict agrees. Without --verify-run a row is labeled self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable). | -| `cost-reduction-90-target` | **hypothesis** | Forgekit reduces coding-agent cost by about 90%. | Cost model (docs/plans/substrate-v2/05-cost-model.md) · plan, corrected 2026-09-26 | [`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`reports/cost-eval.md`](../../reports/cost-eval.md) | The owner's target. The 90.2 / 85.6 / 74.3% scenarios were derived from the refuted 0.62 routing factor and were withdrawn on 2026-09-26; separately estimated stage savings do not multiply into a total. No end-to-end cost has been measured. | -| `phase-p0-specs` | **implemented** | P0: specs 00-08, ADR-0005 and ADR-0006 are merged. | Substrate v2 plan · v0.5.0 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/adr/0005-allow-runtime-dependencies.md`](../../docs/adr/0005-allow-runtime-dependencies.md)
[`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md) | | -| `phase-p1-ledger-core` | **implemented** | P1: the claim ledger stores content-addressed claims whose confidence moves only with independent oracle evidence. | Ledger (src/ledger.js, src/ledger_store.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger.js`](../../src/ledger.js)
[`test/ledger.test.js`](../../test/ledger.test.js) | The review found evidence-semantics defects at the assessed commit: aliases of one git object counted as independent evidence (F06), and a rewritten lesson inherited the old statement's confidence (F07); aedddf5 (committed after the assessed commit) repairs both with regression tests. See ledger-evidence-independence. | -| `phase-p2-team-sync` | **implemented** | P2: ledgers merge across teammates by git union, converging regardless of order. | Ledger sync (src/ledger_sync.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger_sync.js`](../../src/ledger_sync.js)
[`test/ledger_sync.test.js`](../../test/ledger_sync.test.js) | Storage convergence, property-tested; not semantic agreement or factual correctness. The read-path flip (merged legacy + ledger view) and ledger-only writes shipped after P2. | -| `phase-p3-reuse-cache` | **implemented** | P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one. | Reuse cache (src/reuse.js, forge reuse) · master after 1.4.3 (unreleased) | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: the acceptance counterexamples found at d2abfa6, exact keys that erased operators and case (F04) and changed or deleted artifacts that still exact-hit (F05), are repaired with regression tests and a seeded property test. | -| `phase-p4-context-assembly` | **partial** | P4: context assembly never exceeds its token budget and reports a computed missing set. | Context assembly (src/context.js, forge context) · master after 1.4.3 (unreleased) | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/context.js`](../../src/context.js) | Re-assessed on 7eef611: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, and a pointer is not coverage (F02, F03). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. | -| `phase-p5-loop-closure` | **implemented** | P5: outcomes move the confidence of the claims that informed an action; repeated failures mint a diagnosis; imagine dry-runs the selected tests. | Loop closure (src/diagnose.js, src/imagine.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/imagine.js`](../../src/imagine.js)
[`test/imagine.test.js`](../../test/imagine.test.js) | imagine --run is an isolated git checkout of the committed baseline, not a security sandbox, and runs selected files under node --test. | -| `phase-p6-ui-quality-gate` | **implemented** | P6: forge uicheck flags a known-template fixture and passes a project-conformant one, with no LLM calls. | UI checks (src/uicheck.js, forge uicheck) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/plans/substrate-v2/07-ui-quality-gate.md`](../../docs/plans/substrate-v2/07-ui-quality-gate.md)
[`test/uicheck.test.js`](../../test/uicheck.test.js) | The checks are advisory measurements (contrast, token conformance, template distance, fingerprint similarity); none measures accessibility or user value, and none is a blocking hook. | -| `phase-p7-dashboard` | **implemented** | P7: forge dash renders the ledger, cost meter, cache rate and blast radius offline. | Dashboard (src/dash.js, forge dash) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/dash.js`](../../src/dash.js)
[`test/dash.test.js`](../../test/dash.test.js) | The review found read routes answering a foreign Host header at the assessed commit (F13), repaired in aedddf5 (committed after the assessed commit). Cost panels show stage self-estimates, not end-to-end spend. | -| `phase-p8-evaluation` | **partial** | P8: a measured, not asserted, end-to-end cost figure per stage is published in reports/. | Cost evaluation (src/cost_report.js, forge cost --stages) · 1.4.3 | [`reports/cost-eval.md`](../../reports/cost-eval.md)
[`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md) | Stage instrumentation and the stage report exist; no paired end-to-end run has been measured and reports/cost-eval.md holds no data. | -| `context-completeness` | **implemented** | forge context COMPLETE means every required item was delivered as content within the (estimated) token budget. | Context assembly (src/context.js, forge context) · master after 1.4.3 (unreleased) | [`src/context.js`](../../src/context.js)
[`test/context.test.js`](../../test/context.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md) | Re-assessed on 7eef611: F02 (over budget reported as ok) and F03 (a pointer counted as coverage) are regression tests, and a seeded property test holds the rendered block within budget for 50 budgets. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted. | -| `verify-pass-binding` | **implemented** | A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite. | Verification (src/verify.js, forge verify) · master after 1.4.3 (unreleased) | [`src/verify.js`](../../src/verify.js)
[`test/verify.test.js`](../../test/verify.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Re-assessed on 7eef611 (the merge of the 2026-09-26 review fixes): the review's counterexamples F01 (renames, byte moves, empty files, modes, symlinks), F08 (a failing nested workspace) and F10 (code changed during the run) are regression tests, and a seeded property test moves the fingerprint under any composition of manifest transformations. Scope: gitignored files, declared `verify.generated` outputs and interpreter caches are not bound; a nested repository is bound by path only (listed as `unbound`); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted. | -| `ledger-evidence-independence` | **partial** | A claim's confidence rises only with independent evidence for that claim. | Ledger (src/ledger_store.js, src/ledger_bridge.js) · master after 1.4.3 (unreleased) | [`src/ledger.js`](../../src/ledger.js)
[`src/ledger_store.js`](../../src/ledger_store.js)
[`src/ledger_bridge.js`](../../src/ledger_bridge.js)
[`test/ledger_store.test.js`](../../test/ledger_store.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: abbreviations of one git object count as one event, a replayed or re-cited ref cannot refresh decay, and a non-equivalent rewrite no longer inherits trust (F06, F07 regression tests; a seeded property test over spelling mixes, replays and order). Not yet held: two different references to the same underlying run (for example a CI run id and its commit) still count as two events. At d2abfa6 this claim was refuted. | -| `reuse-exact-identity` | **implemented** | An exact reuse hit serves an artifact minted for the same specification (up to whitespace and Unicode normalization), and only while its bytes still match. | Reuse cache (src/reuse.js) · master after 1.4.3 (unreleased) | [`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: F04 (operator, case, literal and polarity pairs) and F05 (edited or deleted artifacts) are regression tests, and a seeded property test never serves an artifact after an edit, move or deletion. Bytes are checked at serve time wherever the repository root is known (`forge reuse query`, the substrate); dependency contracts need an atlas, and without one a hit is marked requiresRevalidation. Artifacts minted before key version 2 never exact-hit. At d2abfa6 this claim was refuted. | -| `imagine-sandbox` | **refuted** | forge imagine --run executes the selected tests in a sandbox. | Consequence simulation (src/imagine.js, forge imagine) · master after 1.4.3 (unreleased) | [`src/imagine.js`](../../src/imagine.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Still false at 7eef611, and no longer claimed: it is an isolated git checkout (a detached-HEAD worktree) that isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. The CLI and docs now say "an isolated checkout of HEAD, not a security sandbox", and a suite that is not node:test gets an explicit unsupported-runner result. | -| `integrations-emission` | **implemented** | Forgekit emits native config for ten coding tools plus MCP config for Roo Code and VS Code. | Config compiler (src/sync.js, src/emit) · 1.4.3 | [`docs/INTEGRATIONS.md`](../../docs/INTEGRATIONS.md)
[`test/sync.test.js`](../../test/sync.test.js)
[`test/mcp.test.js`](../../test/mcp.test.js) | Emission is tested; no host tool is launched in CI. Automatic hooks and in-agent enforcement exist only on Claude Code; a git pre-commit gate covers any tool that commits through git. | -| `pcm-evidence-referenced-memory` | **implemented** | Proof-carrying memory stores content-addressed claims that carry references to their evidence. | Ledger (src/ledger.js) · 1.4.3 | [`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md)
[`src/ledger.js`](../../src/ledger.js) | A name, not a formal proof: there is no theorem prover in the loop. | -| `theorem-d-joint-maxima` | **refuted** | Instructions and deterministic checks together can reach a residual of (1 - p_max)(1 - q_max). | Formal synthesis, Theorem D · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/empirical-refutation/extended_preprint.html`](../../research/empirical-refutation/extended_preprint.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Holds only if both maxima are attainable under one policy. Policy A (0.5, 0.9) leaves 0.05 and policy B (0.9, 0.1) leaves 0.09, while the separate maxima suggest 0.01; the attainable residual is a minimum over the joint feasible set, and the product is a lower bound. Asserted by recompute_corrections.py --theorem-checks. | -| `theorem-d-equality` | **refuted** | Over n independent tasks with per-task residual at most ε, P(at least one miss) equals 1 - (1 - ε)^n. | Formal synthesis, Theorem D · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | It is at most 1 - (1 - ε)^n, with equality only when every per-task residual equals ε; independence alone does not give equality. The union bound nε needs no independence. | -| `independent-checks-400x` | **refuted** | The same check at the Stop hook, pre-commit and CI multiplies as three independent checks. | Formal synthesis, Eq. 5 · HTML edition, corrected 2026-09-21 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Copies of one classifier fire together: the residual is (1-p)(1-c_max) = 0.015, not the product's 3.75e-5 (a 400-fold understatement). Recomputed in recompute_corrections.py section 2 and asserted in section 3b. | -| `silent-miss-implies-completion` | **hypothesis** | A lower silent-miss residual means more completed, correct tasks. | Formal synthesis, section 5.4 · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html) | Not established: a caught mistake can abort, block repeatedly or fail its repair. Measure true catch rate, false-block rate, repaired success given a catch, abandonment, latency and recovery cost. | -| `frozen-model-cannot-adapt` | **refuted** | A frozen model cannot learn, imagine or self-correct, and prompting cannot supply these. | Cognitive-substrate theory (white paper, synthesis) · HTML editions, corrected 2026-09-26 | [`research/cognitive-substrate/cognitive_substrate_whitepaper.html`](../../research/cognitive-substrate/cognitive_substrate_whitepaper.html)
[`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[arxiv.org/abs/2005.14165](https://arxiv.org/abs/2005.14165) | Too broad: examples, retrieved facts and feedback change a frozen model's behaviour within a context with no weight update (Brown et al., 2020). The precise gaps are no durable state across independent invocations, a bounded context, no automatic parameter update, and unreliable self-verification without external evidence. | -| `external-architecture-necessity` | **hypothesis** | An external stateful architecture is necessary to supply the five faculties. | Cognitive-substrate theory · HTML editions, corrected 2026-09-26 | [`research/README.md`](../../research/README.md)
[arxiv.org/abs/2309.02427](https://arxiv.org/abs/2309.02427)
[arxiv.org/abs/2303.11366](https://arxiv.org/abs/2303.11366) | The substrate is one tested way of supplying persistence and verification, not the only logically possible architecture; the five faculties are a decomposition. Prior art: CoALA, Reflexion. Defensible framing: a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes. | -| `independent-convergence` | **refuted** | Four independent arrivals (the theory, forgekit, hikmah-stack and wisdom-lens) confirm one design law. | Formal synthesis, section 14 · HTML edition, corrected 2026-09-21 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/formal-synthesis/README.md`](../../research/formal-synthesis/README.md) | All four share an author: their agreement is consistency, not independent evidence. | -| `metr-19-percent-slowdown` | **reported** | In METR's 2025 trial, 16 experienced open-source developers took 19% longer on 246 tasks with early-2025 AI tools. | External evidence (METR, arXiv:2507.09089) · 2025 study | [`research/cognitive-substrate/evidence/evidence_map.md`](../../research/cognitive-substrate/evidence/evidence_map.md)
[arxiv.org/abs/2507.09089](https://arxiv.org/abs/2507.09089)
[metr.org/blog/2026-02-24-uplift-update](https://metr.org/blog/2026-02-24-uplift-update/) | A result about that population and tooling, not a universal 2026 productivity coefficient in either direction; METR's February 2026 update explains why selection effects complicate newer estimates. | -| `swebench-verified-audit` | **reported** | OpenAI found flawed tests in 59.4% of an audited 138-problem hard subset of SWE-bench Verified, plus contamination evidence. | External evidence (OpenAI, 2026-02-23) · 2026-02-23 analysis | [openai.com/index/why-we-no-longer-evaluate-swe-bench-verified](https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/)
[`research/cognitive-substrate/evidence/evidence_map.md`](../../research/cognitive-substrate/evidence/evidence_map.md) | Not a finding about 59.4% of all 500 tasks. It limits what any SWE-bench Verified replay, including the universal router's, can show. | +| ID | Status | Assessed | Claim | Component · version | Evidence | Notes | +| --- | --- | --- | --- | --- | --- | --- | +| `impact-oracle-perfect-recall` | **refuted** | 1.4.3 · `d2abfa69` | The Python impact oracle has perfect recall: it never misses an affected module. | Python impact oracle (research/python-prototypes/impact_oracle) · v1, as shipped | [`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/impact_oracle/README.md`](../../research/python-prototypes/impact_oracle/README.md) | Recall 1.00 held on five mutations of a ten-file demo package the authors wrote (precision 0.633, F1 0.753). On 759 evaluated files in nine open-source repositories its recall was 0.022. | +| `impact-oracle-field-original` | **measured** | 1.4.3 · `d2abfa69` | On nine repositories (co-edited-file prediction) the as-shipped oracle scores pooled P/R/F1 0.3982/0.0220/0.0416 against grep's 0.3535/0.5732/0.4373. | Python impact oracle (research/python-prototypes/impact_oracle) · v1, as shipped | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Recomputed by the 2026-09-26 external review: 801 labelled / 759 evaluated files, 20,144 mirrored pairs; repository-cluster bootstrap (20,000 draws, seed 1234) F1 oracle [0.0010, 0.0927], grep [0.3807, 0.5394], grep minus oracle [0.3422, 0.5174]; macro F1 0.0220 vs 0.4947. Co-change is a proxy for historically related edits; regression-test selection was not measured. | +| `impact-oracle-repaired-heldout` | **measured** | 1.4.3 · `d2abfa69` | The repaired oracle's F1 on three held-out repositories is 0.416 against grep's 0.371 at the canonical threshold 0.02. | Python impact oracle (research/python-prototypes/impact_oracle) · v2, repaired (in-tree, 49 tests) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/impact_oracle/README.md`](../../research/python-prototypes/impact_oracle/README.md) | A point estimate (ΔF1 about +0.044). Per-repository counts exist only at threshold 0.10 (pooled ΔF1 +0.0565, 3/3 repositories favour it, one-sided sign test p = 0.125); pytest supplies 71.3% of held-out pairs. Never mix the 0.02 and 0.10 figures. | +| `impact-oracle-repaired-beats-grep` | **hypothesis** | 1.4.3 · `d2abfa69` | The repaired oracle beats a grep baseline on repositories it has not seen. | Python impact oracle (research/python-prototypes/impact_oracle) · v2, repaired | [`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/README.md`](../../research/README.md) | Not established: three held-out repositories, and the choice of which relations to add was made after diagnosing all nine (an architecture-selection channel into the test set). Next study: freeze parser and relation design before acquiring new repositories or a time split; report review-cost metrics beside F1. | +| `impact-oracle-in-tree-is-v2` | **implemented** | 1.4.3 · `d2abfa69` | The in-tree Python impact oracle is the repaired v2, with 49 tests (36 demo-package + 13 repair regressions). | Python impact oracle (research/python-prototypes/impact_oracle) · v2 | [`research/python-prototypes/impact_oracle/tests/test_repair_fixes.py`](../../research/python-prototypes/impact_oracle/tests/test_repair_fixes.py)
[`research/python-prototypes/impact_oracle/tests/test_demo_package.py`](../../research/python-prototypes/impact_oracle/tests/test_demo_package.py) | research/README.md said until 2026-09-26 that the repaired oracle shipped only in the replication archive. ImpactOracle(wm, sibling_enabled=False, forward_enabled=False) reproduces the refuted v1 traversal. | +| `node-impact-fixture-quality` | **measured** | 1.5.0 · `7eef6117` | forge impact scores precision 0.17, recall 1.00, F1 0.28 on six hand-labelled cases from this repository. | Node code graph (src/atlas.js, forge impact) · 1.5.0 | [`reports/benchmarks.md`](../../reports/benchmarks.md)
[`bench/impact_cases.mjs`](../../bench/impact_cases.mjs) | Six self-labelled symbols in one repository; the transitive walk is scored against direct-only labels. Re-measured on the tree merged as 7eef611, after the CLI handler move added src/cli/memory.js as an eleventh claimText dependent (reports/benchmarks.md: mean F1 0.28, was 0.29). The 2026-09-26 review measured macro 0.18 / 1.00 / 0.30 on the same cases at d2abfa6. A regex-derived graph, not the evaluated Python AST oracle; none of the Python study's numbers apply to it. | +| `router-gate-demo-saving` | **refuted** | 1.4.3 · `d2abfa69` | Complexity routing saves 62.1% of cost against always using the premium tier. | Old tiered router (research/python-prototypes/router_gate) · July 2026 prototype, thresholds tuned on 30 tasks | [`research/python-prototypes/router_gate/eval_results.json`](../../research/python-prototypes/router_gate/eval_results.json)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Measured on the 30 tasks its thresholds were tuned on, by repricing measured tokens (a price counterfactual). On 80 held-out tasks its total spend was 20.2% higher. | +| `router-gate-heldout-cost` | **measured** | 1.4.3 · `d2abfa69` | On 64 non-halted held-out tasks the routed pipeline spent $6.3582 against always-premium's $5.2893: 20.21% more. | Old tiered router (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Escalation-inclusive and ungated. 58 of 64 tasks failed at every tier, which makes the inversion largely mechanical. | +| `router-gate-judge-accepted-cost` | **measured** | 1.4.3 · `d2abfa69` | Per judge-accepted output the routed pipeline cost $1.060 against always-premium's $1.763. | Old tiered router (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/router_gate/README.md`](../../research/python-prototypes/router_gate/README.md) | judge_accepted only: 6/64 vs 3/64 (Clopper-Pearson [0.035, 0.193] and [0.010, 0.131]), so the ratio is unstable. The judge is also the mid-tier executor. tests_passed, human_accepted and deployed_without_revert were not measured. | +| `router-gate-gate-f1` | **measured** | 1.4.3 · `d2abfa69` | The assumption gate scores F1 0.37 (recall 0.31) on 80 held-out tasks. | Old tiered router's assumption gate (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/empirical-refutation/extended_preprint.html`](../../research/empirical-refutation/extended_preprint.html) | Against 1.00 on the 30-task tuning set. Labels are one model's. | +| `router-gate-label-agreement` | **measured** | 1.4.3 · `d2abfa69` | Re-labelling 30 held-out tasks gives halt κ 0.5161 and tier κ 0.8919. | Old tiered router evaluation labels · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Same model, reworded prompt: self-consistency, not agreement with a human. | +| `forge-route-tier-cost` | **hypothesis** | 1.4.3 · `d2abfa69` | The tiered forge route recommendation lowers cost at equal success. | Tiered router (src/route.js, forge route) · 1.4.3 | [`src/route.js`](../../src/route.js)
[`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md) | Never evaluated end to end on cost. A separately versioned descendant of the refuted router_gate rubric; no measured routing factor supports a positive saving. | +| `universal-router-heldout-headline` | **reported** | 1.4.3 · `d2abfa69` | On 350 held-out SWE-bench Verified issues the universal router (match-best-single) solves 76.3% at $0.093 per task, against 75.1% at $0.364 for the best single model chosen on dev. | Universal router (src/router, forge route universal) · fit on 150 dev issues (harness-bench run 4), not the shipped prior | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Produced by an external harness (harness-bench) that is not shipped here. Its 150/350 split ids, pre-registration, baseline selection and metric aggregation are not in this repository, so it cannot be reproduced from it. The 2026-09-22 re-run (217 of 218 metric values identical) is the project's own report with the same harness, not an independent replication. bench/universal-router/holdout_eval.mjs is a different experiment and does not stand in for it. | +| `universal-router-vs-fixed-cascade` | **measured** | 1.4.3 · `d2abfa69` | With a 150-task dev fit the universal router does not beat a fixed cascade chosen on the same dev data. | Universal router (src/router, forge route universal) · dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5 | [`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | A new in-repo experiment, not a reproduction of run 4: seeded split 20260926, 150 dev / 350 held-out, fit on dev only, 10,000 paired bootstrap draws. Router 80.0% solved at $0.124 per task; best fixed cascade on dev (minimax-m2.5 > gpt-5-mini > kimi-k2.5) 80.6% at $0.136; router minus cascade -0.6 points [-1.4, 0.0] and -$0.012 [-$0.024, -$0.003]. Five more seeds, run after the first result: -0.6 to +1.1 points, never both more accurate and cheaper. The run-4 version of this finding is repository-reported. | +| `universal-router-inrepo-replay-vs-best-single` | **measured** | 1.4.3 · `d2abfa69` | On a new in-repo split (seed 20260926; 150 dev / 350 held-out tasks), the router solves 80.0% of held-out tasks at $0.124 per task, against 76.0% at $0.768 for the best single model chosen on dev (claude-opus-4.5). | Universal router (src/router, forge route universal) · dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5 | [`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | +4.0 points [+0.6, +7.4] at -$0.644 per task [-$0.698, -$0.594], mostly from one cheap model: minimax-m2.5 opens 308 of the 350 cascades. It depends on which model wins dev: across six splits the solve-rate interval excludes zero only for this seed. The replay assumes a perfect, free check between attempts; one scaffold, twelve Python repositories, February 2026 costs, and SWE-bench Verified's label-validity limits. | +| `universal-router-prior-refit` | **measured** | 1.4.3 · `d2abfa69` | The shipped prior (data/router_prior.json) refits exactly from pinned public data with this repository's own tooling. | Universal router prior (data/router_prior.json) · fitted 2026-09-22, k=1, scale 4 | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/reproduce.sh`](../../bench/universal-router/reproduce.sh)
[`bench/universal-router/sources.json`](../../bench/universal-router/sources.json)
[`bench/universal-router/compare_priors.mjs`](../../bench/universal-router/compare_priors.mjs)
[`bench/universal-router/fit_prior.mjs`](../../bench/universal-router/fit_prior.mjs) | Reproduced on 2026-09-26 by the project in its own environment (Node v22.22.2, Python 3.11.15, 4-vCPU Intel Xeon @ 2.80GHz): from an empty work directory all 176 fitted values are identical, only provenance.fittedAt differs, with the router code of d2abfa6 and again with aedddf5; the refit takes 413-424 s. The external review's refit stopped at a 180 s limit, which was too short; the 37.6 s / 45.9 s in the 2026-09-22 report were the 150-issue dev fit. Tasks: Hugging Face SWE-bench/SWE-bench_Verified at revision 78f471b (not princeton-nlp/SWE-bench_Verified). reproduce.sh and sources.json were added after the assessed commit. A calculation reproduction, not an independent third-party replication. | +| `universal-router-prior-transfer` | **hypothesis** | 1.4.3 · `d2abfa69` | The shipped prior predicts success and cost on a user's own workload. | Universal router prior (data/router_prior.json) · fitted 2026-09-22 | [`data/router_prior.json`](../../data/router_prior.json)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[openai.com/index/why-we-no-longer-evaluate-swe-bench-verified](https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/) | Fitted on all 500 SWE-bench Verified tasks (so it cannot be evaluated out of sample on them), one scaffold, Python repositories, February 2026 costs. OpenAI reports test flaws in 59.4% of an audited 138-problem hard subset (not of all 500) plus contamination evidence. Next: time-held-out or unseen repositories, another scaffold, another language, real spend. | +| `universal-router-target-guarantee` | **refuted** | 1.5.0 · `7eef6117` | The target:p objective guarantees a success rate of at least p. | Universal router (src/router, forge route universal) · 1.5.0 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist. Unchanged at 7eef611. | +| `universal-router-budget-contract` | **partial** | 1.5.0 · `7eef6117` | The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists. | Universal router (src/router, forge route universal) · 1.5.0 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`src/router/policy.js`](../../src/router/policy.js) | B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). The explicit half now holds at 7eef611: an infeasible budget returns ok:false, feasible:false, budgetMet:false, minimumExpectedCost and a reason (the CLI prints INFEASIBLE), with the least-bad cascade only as a labeled fallback, and every recommendation reports maxPossibleCost (review F12, regression-tested). | +| `universal-router-cascade-cost-independence` | **refuted** | 1.5.0 · `7eef6117` | A later cascade attempt costs the same in expectation whether or not earlier attempts failed. | Universal router cost model (src/router/policy.js) · 1.5.0 | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | The cascade cost formula assumes E[cost_i \| earlier attempts failed, x] = E[cost_i \| x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes. Unchanged at 7eef611. | +| `route-outcome-provenance` | **partial** | 1.5.0 · `7eef6117` | Outcomes recorded with forge route outcome are verified results. | Universal router outcome log (src/router/index.js) · 1.5.0 | [`src/router/index.js`](../../src/router/index.js)
[`test/router_universal.test.js`](../../test/router_universal.test.js)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | Re-assessed on 7eef611: records are schema-validated on write and read, recording is idempotent by attemptId (--attempt), and a row is provenance verify-event only when --verify-run names a forge verify run whose verdict agrees. Without --verify-run a row is labeled self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable). | +| `cost-reduction-90-target` | **hypothesis** | 1.4.3 · `d2abfa69` | Forgekit reduces coding-agent cost by about 90%. | Cost model (docs/plans/substrate-v2/05-cost-model.md) · plan, corrected 2026-09-26 | [`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`reports/cost-eval.md`](../../reports/cost-eval.md) | The owner's target. The 90.2 / 85.6 / 74.3% scenarios were derived from the refuted 0.62 routing factor and were withdrawn on 2026-09-26; separately estimated stage savings do not multiply into a total. No end-to-end cost has been measured. | +| `phase-p0-specs` | **implemented** | 1.4.3 · `d2abfa69` | P0: specs 00-08, ADR-0005 and ADR-0006 are merged. | Substrate v2 plan · v0.5.0 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/adr/0005-allow-runtime-dependencies.md`](../../docs/adr/0005-allow-runtime-dependencies.md)
[`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md) | | +| `phase-p1-ledger-core` | **implemented** | 1.4.3 · `d2abfa69` | P1: the claim ledger stores content-addressed claims whose confidence moves only with independent oracle evidence. | Ledger (src/ledger.js, src/ledger_store.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger.js`](../../src/ledger.js)
[`test/ledger.test.js`](../../test/ledger.test.js) | The review found evidence-semantics defects at the assessed commit: aliases of one git object counted as independent evidence (F06), and a rewritten lesson inherited the old statement's confidence (F07); aedddf5 (committed after the assessed commit) repairs both with regression tests. See ledger-evidence-independence. | +| `phase-p2-team-sync` | **implemented** | 1.4.3 · `d2abfa69` | P2: ledgers merge across teammates by git union, converging regardless of order. | Ledger sync (src/ledger_sync.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger_sync.js`](../../src/ledger_sync.js)
[`test/ledger_sync.test.js`](../../test/ledger_sync.test.js) | Storage convergence, property-tested; not semantic agreement or factual correctness. The read-path flip (merged legacy + ledger view) and ledger-only writes shipped after P2. | +| `phase-p3-reuse-cache` | **implemented** | 1.5.0 · `7eef6117` | P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one. | Reuse cache (src/reuse.js, forge reuse) · 1.5.0 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: the acceptance counterexamples found at d2abfa6, exact keys that erased operators and case (F04) and changed or deleted artifacts that still exact-hit (F05), are repaired with regression tests and a seeded property test. | +| `phase-p4-context-assembly` | **partial** | 1.5.0 · `7eef6117` | P4: context assembly never exceeds its token budget and reports a computed missing set. | Context assembly (src/context.js, forge context) · 1.5.0 | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/context.js`](../../src/context.js) | Re-assessed on 7eef611: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, and a pointer is not coverage (F02, F03). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. | +| `phase-p5-loop-closure` | **implemented** | 1.4.3 · `d2abfa69` | P5: outcomes move the confidence of the claims that informed an action; repeated failures mint a diagnosis; imagine dry-runs the selected tests. | Loop closure (src/diagnose.js, src/imagine.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/imagine.js`](../../src/imagine.js)
[`test/imagine.test.js`](../../test/imagine.test.js) | imagine --run is an isolated git checkout of the committed baseline, not a security sandbox, and runs selected files under node --test. | +| `phase-p6-ui-quality-gate` | **implemented** | 1.4.3 · `d2abfa69` | P6: forge uicheck flags a known-template fixture and passes a project-conformant one, with no LLM calls. | UI checks (src/uicheck.js, forge uicheck) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/plans/substrate-v2/07-ui-quality-gate.md`](../../docs/plans/substrate-v2/07-ui-quality-gate.md)
[`test/uicheck.test.js`](../../test/uicheck.test.js) | The checks are advisory measurements (contrast, token conformance, template distance, fingerprint similarity); none measures accessibility or user value, and none is a blocking hook. | +| `phase-p7-dashboard` | **implemented** | 1.4.3 · `d2abfa69` | P7: forge dash renders the ledger, cost meter, cache rate and blast radius offline. | Dashboard (src/dash.js, forge dash) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/dash.js`](../../src/dash.js)
[`test/dash.test.js`](../../test/dash.test.js) | The review found read routes answering a foreign Host header at the assessed commit (F13), repaired in aedddf5 (committed after the assessed commit). Cost panels show stage self-estimates, not end-to-end spend. | +| `phase-p8-evaluation` | **partial** | 1.4.3 · `d2abfa69` | P8: a measured, not asserted, end-to-end cost figure per stage is published in reports/. | Cost evaluation (src/cost_report.js, forge cost --stages) · 1.4.3 | [`reports/cost-eval.md`](../../reports/cost-eval.md)
[`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md) | Stage instrumentation and the stage report exist; no paired end-to-end run has been measured and reports/cost-eval.md holds no data. | +| `context-completeness` | **implemented** | 1.5.0 · `7eef6117` | forge context COMPLETE means every required item was delivered as content within the (estimated) token budget. | Context assembly (src/context.js, forge context) · 1.5.0 | [`src/context.js`](../../src/context.js)
[`test/context.test.js`](../../test/context.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md) | Re-assessed on 7eef611: F02 (over budget reported as ok) and F03 (a pointer counted as coverage) are regression tests, and a seeded property test holds the rendered block within budget for 50 budgets. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted. | +| `verify-pass-binding` | **implemented** | 1.5.0 · `7eef6117` | A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite. | Verification (src/verify.js, forge verify) · 1.5.0 | [`src/verify.js`](../../src/verify.js)
[`test/verify.test.js`](../../test/verify.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Re-assessed on 7eef611 (the merge of the 2026-09-26 review fixes): the review's counterexamples F01 (renames, byte moves, empty files, modes, symlinks), F08 (a failing nested workspace) and F10 (code changed during the run) are regression tests, and a seeded property test moves the fingerprint under any composition of manifest transformations. Scope: gitignored files, declared `verify.generated` outputs and interpreter caches are not bound; a nested repository is bound by path only (listed as `unbound`); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted. | +| `ledger-evidence-independence` | **partial** | 1.5.0 · `7eef6117` | A claim's confidence rises only with independent evidence for that claim. | Ledger (src/ledger_store.js, src/ledger_bridge.js) · 1.5.0 | [`src/ledger.js`](../../src/ledger.js)
[`src/ledger_store.js`](../../src/ledger_store.js)
[`src/ledger_bridge.js`](../../src/ledger_bridge.js)
[`test/ledger_store.test.js`](../../test/ledger_store.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: abbreviations of one git object count as one event, a replayed or re-cited ref cannot refresh decay, and a non-equivalent rewrite no longer inherits trust (F06, F07 regression tests; a seeded property test over spelling mixes, replays and order). Not yet held: two different references to the same underlying run (for example a CI run id and its commit) still count as two events. At d2abfa6 this claim was refuted. | +| `reuse-exact-identity` | **implemented** | 1.5.0 · `7eef6117` | An exact reuse hit serves an artifact minted for the same specification (up to whitespace and Unicode normalization), and only while its bytes still match. | Reuse cache (src/reuse.js) · 1.5.0 | [`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: F04 (operator, case, literal and polarity pairs) and F05 (edited or deleted artifacts) are regression tests, and a seeded property test never serves an artifact after an edit, move or deletion. Bytes are checked at serve time wherever the repository root is known (`forge reuse query`, the substrate); dependency contracts need an atlas, and without one a hit is marked requiresRevalidation. Artifacts minted before key version 2 never exact-hit. At d2abfa6 this claim was refuted. | +| `imagine-sandbox` | **refuted** | 1.5.0 · `7eef6117` | forge imagine --run executes the selected tests in a sandbox. | Consequence simulation (src/imagine.js, forge imagine) · 1.5.0 | [`src/imagine.js`](../../src/imagine.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Still false at 7eef611, and no longer claimed: it is an isolated git checkout (a detached-HEAD worktree) that isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. The CLI and docs now say "an isolated checkout of HEAD, not a security sandbox", and a suite that is not node:test gets an explicit unsupported-runner result. | +| `integrations-emission` | **implemented** | 1.4.3 · `d2abfa69` | Forgekit emits native config for ten coding tools plus MCP config for Roo Code and VS Code. | Config compiler (src/sync.js, src/emit) · 1.4.3 | [`docs/INTEGRATIONS.md`](../../docs/INTEGRATIONS.md)
[`test/sync.test.js`](../../test/sync.test.js)
[`test/mcp.test.js`](../../test/mcp.test.js) | Emission is tested; no host tool is launched in CI. Automatic hooks and in-agent enforcement exist only on Claude Code; a git pre-commit gate covers any tool that commits through git. | +| `pcm-evidence-referenced-memory` | **implemented** | 1.4.3 · `d2abfa69` | Proof-carrying memory stores content-addressed claims that carry references to their evidence. | Ledger (src/ledger.js) · 1.4.3 | [`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md)
[`src/ledger.js`](../../src/ledger.js) | A name, not a formal proof: there is no theorem prover in the loop. | +| `theorem-d-joint-maxima` | **refuted** | 1.4.3 · `d2abfa69` | Instructions and deterministic checks together can reach a residual of (1 - p_max)(1 - q_max). | Formal synthesis, Theorem D · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/empirical-refutation/extended_preprint.html`](../../research/empirical-refutation/extended_preprint.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Holds only if both maxima are attainable under one policy. Policy A (0.5, 0.9) leaves 0.05 and policy B (0.9, 0.1) leaves 0.09, while the separate maxima suggest 0.01; the attainable residual is a minimum over the joint feasible set, and the product is a lower bound. Asserted by recompute_corrections.py --theorem-checks. | +| `theorem-d-equality` | **refuted** | 1.4.3 · `d2abfa69` | Over n independent tasks with per-task residual at most ε, P(at least one miss) equals 1 - (1 - ε)^n. | Formal synthesis, Theorem D · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | It is at most 1 - (1 - ε)^n, with equality only when every per-task residual equals ε; independence alone does not give equality. The union bound nε needs no independence. | +| `independent-checks-400x` | **refuted** | 1.4.3 · `d2abfa69` | The same check at the Stop hook, pre-commit and CI multiplies as three independent checks. | Formal synthesis, Eq. 5 · HTML edition, corrected 2026-09-21 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Copies of one classifier fire together: the residual is (1-p)(1-c_max) = 0.015, not the product's 3.75e-5 (a 400-fold understatement). Recomputed in recompute_corrections.py section 2 and asserted in section 3b. | +| `silent-miss-implies-completion` | **hypothesis** | 1.4.3 · `d2abfa69` | A lower silent-miss residual means more completed, correct tasks. | Formal synthesis, section 5.4 · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html) | Not established: a caught mistake can abort, block repeatedly or fail its repair. Measure true catch rate, false-block rate, repaired success given a catch, abandonment, latency and recovery cost. | +| `frozen-model-cannot-adapt` | **refuted** | 1.4.3 · `d2abfa69` | A frozen model cannot learn, imagine or self-correct, and prompting cannot supply these. | Cognitive-substrate theory (white paper, synthesis) · HTML editions, corrected 2026-09-26 | [`research/cognitive-substrate/cognitive_substrate_whitepaper.html`](../../research/cognitive-substrate/cognitive_substrate_whitepaper.html)
[`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[arxiv.org/abs/2005.14165](https://arxiv.org/abs/2005.14165) | Too broad: examples, retrieved facts and feedback change a frozen model's behaviour within a context with no weight update (Brown et al., 2020). The precise gaps are no durable state across independent invocations, a bounded context, no automatic parameter update, and unreliable self-verification without external evidence. | +| `external-architecture-necessity` | **hypothesis** | 1.4.3 · `d2abfa69` | An external stateful architecture is necessary to supply the five faculties. | Cognitive-substrate theory · HTML editions, corrected 2026-09-26 | [`research/README.md`](../../research/README.md)
[arxiv.org/abs/2309.02427](https://arxiv.org/abs/2309.02427)
[arxiv.org/abs/2303.11366](https://arxiv.org/abs/2303.11366) | The substrate is one tested way of supplying persistence and verification, not the only logically possible architecture; the five faculties are a decomposition. Prior art: CoALA, Reflexion. Defensible framing: a portable implementation of evidence-weighted coding-agent memory and checks, with empirical evaluation of trust failure modes. | +| `independent-convergence` | **refuted** | 1.4.3 · `d2abfa69` | Four independent arrivals (the theory, forgekit, hikmah-stack and wisdom-lens) confirm one design law. | Formal synthesis, section 14 · HTML edition, corrected 2026-09-21 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/formal-synthesis/README.md`](../../research/formal-synthesis/README.md) | All four share an author: their agreement is consistency, not independent evidence. | +| `metr-19-percent-slowdown` | **reported** | 1.4.3 · `d2abfa69` | In METR's 2025 trial, 16 experienced open-source developers took 19% longer on 246 tasks with early-2025 AI tools. | External evidence (METR, arXiv:2507.09089) · 2025 study | [`research/cognitive-substrate/evidence/evidence_map.md`](../../research/cognitive-substrate/evidence/evidence_map.md)
[arxiv.org/abs/2507.09089](https://arxiv.org/abs/2507.09089)
[metr.org/blog/2026-02-24-uplift-update](https://metr.org/blog/2026-02-24-uplift-update/) | A result about that population and tooling, not a universal 2026 productivity coefficient in either direction; METR's February 2026 update explains why selection effects complicate newer estimates. | +| `swebench-verified-audit` | **reported** | 1.4.3 · `d2abfa69` | OpenAI found flawed tests in 59.4% of an audited 138-problem hard subset of SWE-bench Verified, plus contamination evidence. | External evidence (OpenAI, 2026-02-23) · 2026-02-23 analysis | [openai.com/index/why-we-no-longer-evaluate-swe-bench-verified](https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/)
[`research/cognitive-substrate/evidence/evidence_map.md`](../../research/cognitive-substrate/evidence/evidence_map.md) | Not a finding about 59.4% of all 500 tasks. It limits what any SWE-bench Verified replay, including the universal router's, can show. | diff --git a/docs/status/claims.json b/docs/status/claims.json index 02fcb742..607f2b0e 100644 --- a/docs/status/claims.json +++ b/docs/status/claims.json @@ -1,5 +1,5 @@ { - "$comment": "The claim/status registry. Edit this file, then run `node scripts/claims-status.mjs` to regenerate the table in docs/status/README.md; `--check` fails CI-style on drift. source_commit is the commit each status was assessed against; after a fix lands, re-assess the claim and update both status and source_commit.", + "$comment": "The claim/status registry. Edit this file, then run `node scripts/claims-status.mjs` to regenerate the table in docs/status/README.md; `--check` fails CI-style on drift. source_commit is the commit each status was assessed against and assessed_release the release that ships that commit (\"unreleased\" until one does: scripts/bump.mjs stamps it when it cuts a release, and a checkout with release tags fails the check on a stale one). counterexamples link the review findings that tested a claim. After a fix lands, re-assess the claim and update status, source_commit, assessed_release and the counterexample's resolution together.", "as_of": "2026-09-26", "statuses": { "implemented": "the code exists and does what the claim says; tests exercise it", @@ -16,6 +16,7 @@ "component": "Python impact oracle (research/python-prototypes/impact_oracle)", "version": "v1, as shipped", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/empirical-refutation/README.md", @@ -30,6 +31,7 @@ "component": "Python impact oracle (research/python-prototypes/impact_oracle)", "version": "v1, as shipped", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "research/empirical-refutation/replication_package.tar.gz", @@ -44,6 +46,7 @@ "component": "Python impact oracle (research/python-prototypes/impact_oracle)", "version": "v2, repaired (in-tree, 49 tests)", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "research/empirical-refutation/replication_package.tar.gz", @@ -58,6 +61,7 @@ "component": "Python impact oracle (research/python-prototypes/impact_oracle)", "version": "v2, repaired", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "hypothesis", "evidence": [ "research/empirical-refutation/README.md", @@ -71,6 +75,7 @@ "component": "Python impact oracle (research/python-prototypes/impact_oracle)", "version": "v2", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "research/python-prototypes/impact_oracle/tests/test_repair_fixes.py", @@ -82,8 +87,9 @@ "id": "node-impact-fixture-quality", "claim": "forge impact scores precision 0.17, recall 1.00, F1 0.28 on six hand-labelled cases from this repository.", "component": "Node code graph (src/atlas.js, forge impact)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "measured", "evidence": [ "reports/benchmarks.md", @@ -97,6 +103,7 @@ "component": "Old tiered router (research/python-prototypes/router_gate)", "version": "July 2026 prototype, thresholds tuned on 30 tasks", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/python-prototypes/router_gate/eval_results.json", @@ -110,6 +117,7 @@ "component": "Old tiered router (research/python-prototypes/router_gate)", "version": "as evaluated (replication archive)", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "research/empirical-refutation/replication_package.tar.gz", @@ -124,6 +132,7 @@ "component": "Old tiered router (research/python-prototypes/router_gate)", "version": "as evaluated (replication archive)", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "research/empirical-refutation/replication_package.tar.gz", @@ -138,6 +147,7 @@ "component": "Old tiered router's assumption gate (research/python-prototypes/router_gate)", "version": "as evaluated (replication archive)", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "research/empirical-refutation/replication_package.tar.gz", @@ -152,6 +162,7 @@ "component": "Old tiered router evaluation labels", "version": "as evaluated (replication archive)", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "research/empirical-refutation/replication_package.tar.gz", @@ -165,6 +176,7 @@ "component": "Tiered router (src/route.js, forge route)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "hypothesis", "evidence": [ "src/route.js", @@ -178,6 +190,7 @@ "component": "Universal router (src/router, forge route universal)", "version": "fit on 150 dev issues (harness-bench run 4), not the shipped prior", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "reported", "evidence": [ "docs/UNIVERSAL_ROUTING.md", @@ -191,6 +204,7 @@ "component": "Universal router (src/router, forge route universal)", "version": "dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "bench/universal-router/holdout_eval.mjs", @@ -205,6 +219,7 @@ "component": "Universal router (src/router, forge route universal)", "version": "dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "bench/universal-router/holdout_eval.mjs", @@ -219,6 +234,7 @@ "component": "Universal router prior (data/router_prior.json)", "version": "fitted 2026-09-22, k=1, scale 4", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "measured", "evidence": [ "bench/universal-router/README.md", @@ -235,6 +251,7 @@ "component": "Universal router prior (data/router_prior.json)", "version": "fitted 2026-09-22", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "hypothesis", "evidence": [ "data/router_prior.json", @@ -247,8 +264,9 @@ "id": "universal-router-target-guarantee", "claim": "The target:p objective guarantees a success rate of at least p.", "component": "Universal router (src/router, forge route universal)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "refuted", "evidence": [ "docs/UNIVERSAL_ROUTING.md", @@ -260,8 +278,9 @@ "id": "universal-router-budget-contract", "claim": "The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists.", "component": "Universal router (src/router, forge route universal)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "partial", "evidence": [ "docs/UNIVERSAL_ROUTING.md", @@ -273,8 +292,9 @@ "id": "universal-router-cascade-cost-independence", "claim": "A later cascade attempt costs the same in expectation whether or not earlier attempts failed.", "component": "Universal router cost model (src/router/policy.js)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "refuted", "evidence": [ "bench/universal-router/README.md", @@ -287,8 +307,9 @@ "id": "route-outcome-provenance", "claim": "Outcomes recorded with forge route outcome are verified results.", "component": "Universal router outcome log (src/router/index.js)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "partial", "evidence": [ "src/router/index.js", @@ -303,6 +324,7 @@ "component": "Cost model (docs/plans/substrate-v2/05-cost-model.md)", "version": "plan, corrected 2026-09-26", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "hypothesis", "evidence": [ "docs/plans/substrate-v2/05-cost-model.md", @@ -316,6 +338,7 @@ "component": "Substrate v2 plan", "version": "v0.5.0", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -330,6 +353,7 @@ "component": "Ledger (src/ledger.js, src/ledger_store.js)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -344,6 +368,7 @@ "component": "Ledger sync (src/ledger_sync.js)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -356,8 +381,9 @@ "id": "phase-p3-reuse-cache", "claim": "P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one.", "component": "Reuse cache (src/reuse.js, forge reuse)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -371,8 +397,9 @@ "id": "phase-p4-context-assembly", "claim": "P4: context assembly never exceeds its token budget and reports a computed missing set.", "component": "Context assembly (src/context.js, forge context)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "partial", "evidence": [ "docs/plans/substrate-v2/04-context-assembly.md", @@ -387,6 +414,7 @@ "component": "Loop closure (src/diagnose.js, src/imagine.js)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -401,6 +429,7 @@ "component": "UI checks (src/uicheck.js, forge uicheck)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -415,6 +444,7 @@ "component": "Dashboard (src/dash.js, forge dash)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", @@ -429,6 +459,7 @@ "component": "Cost evaluation (src/cost_report.js, forge cost --stages)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "partial", "evidence": [ "reports/cost-eval.md", @@ -441,8 +472,9 @@ "id": "context-completeness", "claim": "forge context COMPLETE means every required item was delivered as content within the (estimated) token budget.", "component": "Context assembly (src/context.js, forge context)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "implemented", "evidence": [ "src/context.js", @@ -456,8 +488,9 @@ "id": "verify-pass-binding", "claim": "A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite.", "component": "Verification (src/verify.js, forge verify)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "implemented", "evidence": [ "src/verify.js", @@ -471,8 +504,9 @@ "id": "ledger-evidence-independence", "claim": "A claim's confidence rises only with independent evidence for that claim.", "component": "Ledger (src/ledger_store.js, src/ledger_bridge.js)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "partial", "evidence": [ "src/ledger.js", @@ -487,8 +521,9 @@ "id": "reuse-exact-identity", "claim": "An exact reuse hit serves an artifact minted for the same specification (up to whitespace and Unicode normalization), and only while its bytes still match.", "component": "Reuse cache (src/reuse.js)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "implemented", "evidence": [ "src/reuse.js", @@ -501,8 +536,9 @@ "id": "imagine-sandbox", "claim": "forge imagine --run executes the selected tests in a sandbox.", "component": "Consequence simulation (src/imagine.js, forge imagine)", - "version": "master after 1.4.3 (unreleased)", + "version": "1.5.0", "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "assessed_release": "1.5.0", "status": "refuted", "evidence": [ "src/imagine.js", @@ -516,6 +552,7 @@ "component": "Config compiler (src/sync.js, src/emit)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/INTEGRATIONS.md", @@ -530,6 +567,7 @@ "component": "Ledger (src/ledger.js)", "version": "1.4.3", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "implemented", "evidence": [ "docs/adr/0006-proof-carrying-memory.md", @@ -543,6 +581,7 @@ "component": "Formal synthesis, Theorem D", "version": "HTML edition, corrected 2026-09-26", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/formal-synthesis/substrate_synthesis.html", @@ -557,6 +596,7 @@ "component": "Formal synthesis, Theorem D", "version": "HTML edition, corrected 2026-09-26", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/formal-synthesis/substrate_synthesis.html", @@ -570,6 +610,7 @@ "component": "Formal synthesis, Eq. 5", "version": "HTML edition, corrected 2026-09-21", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/formal-synthesis/substrate_synthesis.html", @@ -583,6 +624,7 @@ "component": "Formal synthesis, section 5.4", "version": "HTML edition, corrected 2026-09-26", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "hypothesis", "evidence": [ "research/formal-synthesis/substrate_synthesis.html" @@ -595,6 +637,7 @@ "component": "Cognitive-substrate theory (white paper, synthesis)", "version": "HTML editions, corrected 2026-09-26", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/cognitive-substrate/cognitive_substrate_whitepaper.html", @@ -609,6 +652,7 @@ "component": "Cognitive-substrate theory", "version": "HTML editions, corrected 2026-09-26", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "hypothesis", "evidence": [ "research/README.md", @@ -623,6 +667,7 @@ "component": "Formal synthesis, section 14", "version": "HTML edition, corrected 2026-09-21", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "refuted", "evidence": [ "research/formal-synthesis/substrate_synthesis.html", @@ -636,6 +681,7 @@ "component": "External evidence (METR, arXiv:2507.09089)", "version": "2025 study", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "reported", "evidence": [ "research/cognitive-substrate/evidence/evidence_map.md", @@ -650,6 +696,7 @@ "component": "External evidence (OpenAI, 2026-02-23)", "version": "2026-02-23 analysis", "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "assessed_release": "1.4.3", "status": "reported", "evidence": [ "https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/", diff --git a/mintlify/changelog/overview.mdx b/mintlify/changelog/overview.mdx index 406e1c25..792b5b7e 100644 --- a/mintlify/changelog/overview.mdx +++ b/mintlify/changelog/overview.mdx @@ -18,6 +18,38 @@ This page is generated from `CHANGELOG.md` by `forge docs render`, and `forge do CI when it falls behind, so it cannot drift from the release notes again. {/* forge:render:changelog:begin (generated by `forge docs render` — do not edit) */} + + +**Fixed** + +- **Node's `-r` preload flag no longer passes as a recursive workspace run (N03).** +- **Exact reuse is byte-exact (N01).** +- **A dependency's whole declaration is its contract (N08).** +- **Similar rules are never merged or dropped (N02).** +- **An unsigned verifier event never earns verified provenance (N06).** +- **An edited provenance label counts for nothing (N07).** +- **A long function is delivered whole or not at all (N04).** +- **Code in a nested repository is bound by the fingerprint (N05).** + +**Added** + +- **`verify.external`** +- **Coverage basis in `forge verify`.** +- **Verifier event contract v2.** +- **The claim registry is a release artifact.** +- **Property tests along semantic boundaries.** + +**Changed** + +- **`maxPossibleCost` is now `estimatedCostIfAllAttemptsRun`.** +- **Formats that changed rebuild or re-verify on their own.** +- **The semantic guard compares code layout and stops folding Unicode.** +- **The research executive summary marks its 2026-09-26 correction in place.** + +[Full notes for Unreleased →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#unreleased) + + + **Changed** diff --git a/mintlify/cli/memory.mdx b/mintlify/cli/memory.mdx index 1f9c792f..84b10508 100644 --- a/mintlify/cli/memory.mdx +++ b/mintlify/cli/memory.mdx @@ -70,10 +70,12 @@ Evidence counts once per event: four spellings of one commit (`git:` plus 7, 8, characters) are one vote, and a resolvable abbreviation is stored under the full id. A reworded lesson inherits its predecessor's evidence only when the rewrite is equivalent up to case, whitespace and punctuation; otherwise it starts at the prior and records what it -supersedes. `compact` merges near-duplicates only when they also agree on operators, -numbers, literals, identifiers, paths and negation. "Enable X" and "Disable X" are listed as -conflicts for a human to resolve, never merged. Every archived claim records why it went to -the attic (`tombstoned`, `dormant`, `idle` or `duplicate`); archived is not refuted. +supersedes. `compact` archives a duplicate only when it makes the same statement (equal up +to whitespace and trailing punctuation). Near-duplicates are listed for a human to merge, +never archived: similarity cannot tell "allow admins, deny guests" from "deny admins, allow +guests". "Enable X" and "Disable X" are listed as conflicts. Every archived claim records +why it went to the attic (`tombstoned`, `dormant`, `idle` or `duplicate`); archived is not +refuted. ## `forge reuse` @@ -85,11 +87,13 @@ forge reuse mint "" --file # add an artifact to the cache forge reuse stats # cache stats ``` -- **Exact means the same text.** The key ignores only whitespace; case, operators, - literals and punctuation all count, so `age >= 18` and `age <= 18` never share a key. +- **Exact means the same text, byte for byte.** Nothing is normalized: whitespace inside + `"a b"`, indentation and Unicode code points all count, as do case, operators and + punctuation, so `age >= 18` and `age <= 18` never share a key. Inline code that the + ledger's storage would rewrite (CRLF, non-NFC text) is refused at mint: use `--file`. - **Near must also agree on behaviour.** A reworded match whose operators, numbers, - literals, identifiers, paths or negation differ drops to `adapt`, with a note naming the - difference. + literals, identifiers, paths, negation or code layout differ drops to `adapt`, with a + note naming the difference. - **Checked where it is served.** An artifact whose file changed or was deleted since it was minted, or whose dependency's declaration changed, is not served. Without an atlas to check against, a hit is marked `NOT revalidated` (`requiresRevalidation: true`). diff --git a/mintlify/cli/quality.mdx b/mintlify/cli/quality.mdx index 37e5098c..8a9e91d3 100644 --- a/mintlify/cli/quality.mdx +++ b/mintlify/cli/quality.mdx @@ -33,19 +33,25 @@ tests (the root, plus each nested package with an explicit `scripts.test`, a pyt a `go.mod`, …) and runs each in its own directory. `packages: n/m covered` counts the ones that reached a verdict. A package whose suite never reached one makes the result `INCOMPLETE`, and a failing package makes it `FAIL`, however green the root is. A root -script that already runs every workspace (`npm test --workspaces`, `pnpm -r test`, -`turbo run test`, …) covers them in one run. Tune it under `verify` in -`.forge/forge.config.json`: +script covers the workspaces in one run only when forge establishes, from its shell +structure, an unfiltered recursive run of every member's `test` script whose failure reaches +the exit status (`npm test --workspaces`, `pnpm -r test`, `turbo run test`, …). Filters, +masked runs and look-alike flags such as Node's preload `node -r` cover nothing, and each +package then runs its own suite. `coverage basis` labels every verdict `measured`, +`inferred` or `declared`. Tune it under `verify` in `.forge/forge.config.json`: ```json -{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"] } } +{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"], "external": ["vendor/upstream"] } } ``` **Bound to the code that was tested.** The stamp is bound to a fingerprint of the working tree (HEAD, staged and unstaged diffs, and each untracked file's path, mode, size and content hash), taken before **and** after the run. If the code changed while the tests ran (a formatter, a generator, another agent), the result is `INCOMPLETE` with -`mutated: true`. Interpreter caches and the `generated` paths never count as a change. +`mutated: true`. Nested repositories and submodules are bound by their own state, so a test +that rewrites code it imports from one is caught. Code the fingerprint cannot bind makes the +result `INCOMPLETE` unless the repo declares it under `verify.external`. Interpreter caches +and the `generated` paths never count as a change. Each run also appends a MAC-sealed event to `.forge/verify-events.jsonl` (run id, verifier version, suites, coverage, pre/post state), which `forge route outcome --verify-run ` can cite. diff --git a/mintlify/cli/substrate.mdx b/mintlify/cli/substrate.mdx index 1bc0936e..7d4f6511 100644 --- a/mintlify/cli/substrate.mdx +++ b/mintlify/cli/substrate.mdx @@ -103,7 +103,9 @@ reports the computed missing set. separators included), not a tokenizer's count, and never exceed `--budget` while the result reports success. - **A pointer is not coverage.** An item that only fits as a `- read ` pointer is a - `pending` read; a 25-line head covers a definition only when the definition is inside it. + `pending` read. A definition counts as delivered only whole, declaration through its last + line: a span or 25-line head that cuts the body leaves it pending and names it under + `partial` (lines shown vs lines spanned). - **Over budget is `INCOMPLETE`.** When even pointers do not fit, the result is `overflow: true`, `ok: false`, never a silent over-budget pass. diff --git a/mintlify/concepts/model-routing.mdx b/mintlify/concepts/model-routing.mdx index b224c91d..46aa83db 100644 --- a/mintlify/concepts/model-routing.mdx +++ b/mintlify/concepts/model-routing.mdx @@ -126,9 +126,11 @@ forge route universal "" --provider anthropic # only models you can cal `--provider`. - **Learning from your outcomes.** `forge route outcome` records one attempt, validated (known model, finite non-negative cost, the right feature length). It is self-reported - unless `--verify-run ` ties it to a matching `forge verify` run, and - `--attempt ` makes re-recording idempotent. `forge route fit` then updates the shipped - prior with them. + unless `--verify-run ` ties it to an authenticated `forge verify` run with the same + verdict, and one run backs one attempt. The label is re-derived from the verifier events on + every read, so an edited label counts for nothing, and the event vouches for pass/fail + only (never the model or the cost). `--attempt ` makes re-recording idempotent. + `forge route fit` then updates the shipped prior with them. The shipped prior was fitted on public SWE-bench Verified outcomes: one agent scaffold, diff --git a/mintlify/concepts/proof-carrying-memory.mdx b/mintlify/concepts/proof-carrying-memory.mdx index ac8eb3e9..8ee2aaaf 100644 --- a/mintlify/concepts/proof-carrying-memory.mdx +++ b/mintlify/concepts/proof-carrying-memory.mdx @@ -104,9 +104,10 @@ Add `--personal` for the per-user ledger. when its evidence still holds. Its confidence must be above the floor, its file must be byte-identical to what was verified, and its dependencies must still resolve with the same declarations. Otherwise it falls through to generation and mints a fresh claim on the way -back. An exact hit needs the same text (only whitespace is ignored). A near hit must also -agree on operators, numbers, literals, identifiers, paths and negation, so `age >= 18` is -never served for `age <= 18`. +back. An exact hit needs the same text, byte for byte: no whitespace or Unicode +normalization, because the spaces in `"a b"` or a Python block's indentation are data. A +near hit must also agree on operators, numbers, literals, identifiers, paths, negation and +code layout, so `age >= 18` is never served for `age <= 18`. One event is one vote. Abbreviations of one commit count once. A reworded lesson inherits trust only when the rewrite is equivalent. Similar-but-opposite rules are reported as diff --git a/mintlify/concepts/verification-gates.mdx b/mintlify/concepts/verification-gates.mdx index 9ece210c..24421927 100644 --- a/mintlify/concepts/verification-gates.mdx +++ b/mintlify/concepts/verification-gates.mdx @@ -60,22 +60,29 @@ tests (the root, plus each nested package with an explicit `scripts.test`, a pyt a `go.mod`, …) and runs each in its own directory. `packages: n/m covered` counts the ones that reached a verdict. A package whose suite never reached one makes the result `INCOMPLETE`, and a failing package makes it `FAIL`, however green the root is. A root -script that already runs every workspace (`npm test --workspaces`, `pnpm -r test`, -`turbo run test`, …) covers them in one run. Tune it under `verify` in -`.forge/forge.config.json`: +script covers the workspaces in one run only when forge establishes, from its shell +structure, an unfiltered recursive run of every member's `test` script whose failure reaches +the exit status (`npm test --workspaces`, `pnpm -r test`, `turbo run test`, …). Filters, +masked runs and look-alike flags such as Node's preload `node -r` cover nothing, and each +package then runs its own suite. `coverage basis` labels every verdict `measured`, +`inferred` or `declared`. Tune it under `verify` in `.forge/forge.config.json`: ```json -{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"] } } +{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"], "external": ["vendor/upstream"] } } ``` **Bound to the code that was tested.** The stamp is bound to a fingerprint of the working tree (HEAD, staged and unstaged diffs, and each untracked file's path, mode, size and content hash), taken before **and** after the run. If the code changed while the tests ran (a formatter, a generator, another agent), the result is `INCOMPLETE` with -`mutated: true`. Interpreter caches and the `generated` paths never count as a change. -Each run also appends a MAC-sealed event to `.forge/verify-events.jsonl` (run id, verifier -version, suites, coverage, pre/post state), which `forge route outcome --verify-run ` -can cite. +`mutated: true`. Nested repositories and submodules are bound by their own state, so a test +that rewrites code it imports from one is caught. Code the fingerprint cannot bind makes the +result `INCOMPLETE` unless the repo declares it under `verify.external`. Interpreter caches +and the `generated` paths never count as a change. +Each run also appends an event to `.forge/verify-events.jsonl` (run id, verifier version, +suites, coverage, pre/post state, environment). Its machine-local MAC covers every field, so +an edited event reads back as unauthenticated. Without an evidence key nothing is +authenticated. `forge route outcome --verify-run ` can cite an authenticated event. ## The hallucinated-symbol flag — `forge atlas has` diff --git a/research/cognitive-substrate/EXECUTIVE_SUMMARY.md b/research/cognitive-substrate/EXECUTIVE_SUMMARY.md index f6d12774..a848ddce 100644 --- a/research/cognitive-substrate/EXECUTIVE_SUMMARY.md +++ b/research/cognitive-substrate/EXECUTIVE_SUMMARY.md @@ -17,13 +17,24 @@ > *Corrected 2026-09-21* after an external review: the repair was previously called "a narrow win", > and the whitepaper's HTML edition now also marks each refuted claim in place. The whitepaper PDF > predates those corrections. +> +> *Corrected 2026-09-26*: the thesis below, that a frozen model "cannot" remember, learn, +> imagine or self-correct and that prompting or tools "cannot" supply these, is broader than +> the gaps it rests on. Examples, retrieved facts and feedback do change a frozen model's +> behaviour within a context. What it lacks is durable state across independent invocations, +> an unbounded context, automatic parameter updates, and reliable self-verification without +> external evidence. The substrate is one tested way to supply persistence and verification, +> not the only possible one; CoALA and Reflexion are prior art. The claim registry +> ([`docs/status/README.md`](../../docs/status/README.md)) tracks the first statement as refuted +> (`frozen-model-cannot-adapt`) and the necessity of this architecture as a hypothesis +> (`external-architecture-necessity`). --- # A Cognitive Substrate for Coding Agents — Deliverable Package ### Theory → Evidence → Build-Map edition (v2) -**One-line thesis:** The faculties a coding agent lacks — memory, learning, imagination, self-correction, impact-awareness — are not gaps in the model's *knowledge* but structural consequences of what a frozen transformer *is* (a stateless map `y = f_θ(x)`, fixed weights, bounded window). They cannot be prompted or tooled away; they can only be supplied by **re-wrapping the input→process→output loop** into a closed, stateful cycle around the frozen model. +**One-line thesis:** The faculties a coding agent lacks — memory, learning, imagination, self-correction, impact-awareness — are not gaps in the model's *knowledge* but structural consequences of what a frozen transformer *is* (a stateless map `y = f_θ(x)`, fixed weights, bounded window). They cannot be prompted or tooled away; they can only be supplied by **re-wrapping the input→process→output loop** into a closed, stateful cycle around the frozen model. *[Corrected 2026-09-26: too broad as worded; see the status note above.]* > *Corrected 2026-09-26:* the thesis above is broader than its argument. Frozen weights rule out > weight updates during use, not all adaptation: examples, retrieved facts and feedback in the @@ -45,7 +56,7 @@ ## What's in this package ### 1. The white paper (core deliverable) — 48 pp -- **`cognitive_substrate_whitepaper.pdf`** / **`cognitive_substrate_whitepaper.html`** — the full study, 13 sections + 3 appendices, 7 figures. +- **`cognitive_substrate_whitepaper.pdf`** (historical, pre-correction edition) / **`cognitive_substrate_whitepaper.html`** (corrected in place) — the full study, 13 sections + 3 appendices, 7 figures. - **§1–3** the root cause and the five faculties (from v1): *why* each faculty is structurally absent (P1 statelessness, P2 frozen weights, P3 bounded context), each grounded in the real literature. - **§4 Evidence** *(new)* — the twelve statistics, re-grounded. 5 confirmed, 5 vendor-reported, 2 unverifiable. - **§5** the Qur'anic epistemic lens — design framing/ethics, never technical authority. diff --git a/scripts/bump.mjs b/scripts/bump.mjs index 0ffe5ad5..9907a9d9 100644 --- a/scripts/bump.mjs +++ b/scripts/bump.mjs @@ -23,6 +23,8 @@ * CITATION.cff (version + date-released), landing/index.html (display string), * ROADMAP.md ("## Now" marker version, so `forge docs check`'s roadmap-freshness * guard never trails a release this same script just cut), + * docs/status/claims.json (every claim assessed on "unreleased" code is stamped with the + * release that ships it — the registry is a release artifact) + its generated table, * CHANGELOG.md ([Unreleased] rotated under "## [X.Y.Z] - " + compare links). * * Prints ONLY the new version on stdout (diagnostics go to stderr) so callers can @@ -34,6 +36,13 @@ import path from "node:path"; import { fileURLToPath } from "node:url"; import { CHANGELOG_PAGE } from "../src/changelog_page.js"; import { renderFile } from "../src/docs_render.js"; +import { + README_PATH, + REGISTRY_PATH, + renderTable, + spliceReadme, + stampRelease, +} from "./claims-status.mjs"; // --------------------------------------------------------------------------- // Pure version math @@ -354,6 +363,21 @@ export function applyBump(root, currentVersion, newVersion, date) { if (updated !== roadmap) write(roadmapRel, updated); } + // The claim registry is a release artifact (review suggestion 5): claims assessed on + // unreleased code now name the release that ships them, and the status table follows. + const claims = readIfExists(path.join(root, REGISTRY_PATH)); + if (claims !== null) { + const stamped = stampRelease(claims, newVersion); + if (stamped !== claims) { + write(REGISTRY_PATH, stamped); + const readme = readIfExists(path.join(root, README_PATH)); + if (readme !== null) { + const next = spliceReadme(readme.replace(/\r\n/g, "\n"), renderTable(JSON.parse(stamped))); + if (next !== readme) write(README_PATH, next); + } + } + } + const clRel = "CHANGELOG.md"; const changelog = readIfExists(path.join(root, clRel)); if (changelog !== null) { diff --git a/scripts/claims-status.mjs b/scripts/claims-status.mjs index b911ae55..ddc597c5 100644 --- a/scripts/claims-status.mjs +++ b/scripts/claims-status.mjs @@ -3,11 +3,17 @@ * The claim/status registry, checked and rendered (node stdlib only). * * `docs/status/claims.json` records every load-bearing headline the project makes — what is - * claimed, which component and version it is about, the commit it was assessed against, its - * status, and the evidence behind it. This script keeps three things honest: + * claimed, which component and version it is about, the commit it was assessed against, the + * RELEASE that assessment belongs to, its status, the evidence behind it, and the review + * counterexamples that tested it. This script keeps three things honest: * * 1. the registry itself: required fields, a closed set of statuses, unique ids, a commit - * that looks like one, and evidence paths that exist in the repository; + * that looks like one, an assessed release that is a version or "unreleased", well-formed + * counterexample links, and evidence paths that exist in the repository — and, when the + * checkout has release tags, that every assessed release really contains the assessed + * commit and no claim still says "unreleased" about a commit that has shipped (the + * registry is a RELEASE artifact: `scripts/bump.mjs` stamps "unreleased" claims with the + * version it cuts); * 2. the status table in `docs/status/README.md`, which is generated from the registry * between the CLAIMS:BEGIN / CLAIMS:END markers and never edited by hand; * 3. the research copies under `docs/cognitive-substrate/`, which must stay byte-identical @@ -21,6 +27,7 @@ * * Exit codes: 0 ok · 1 invalid registry or drift · 2 usage error. */ +import { execFileSync } from "node:child_process"; import { createHash } from "node:crypto"; import { copyFileSync, existsSync, readFileSync, writeFileSync } from "node:fs"; import path from "node:path"; @@ -36,11 +43,15 @@ export const REQUIRED_FIELDS = [ "component", "version", "source_commit", + "assessed_release", "status", "evidence", "notes", ]; +/** How a review counterexample linked to a claim was resolved. */ +export const RESOLUTIONS = ["fixed", "open", "scoped"]; + export const REGISTRY_PATH = "docs/status/claims.json"; export const README_PATH = "docs/status/README.md"; export const BEGIN = ""; @@ -73,6 +84,8 @@ export const COPY_PAIRS = [ const ID_RE = /^[a-z0-9][a-z0-9-]*$/; const COMMIT_RE = /^[0-9a-f]{7,40}$/; const URL_RE = /^https?:\/\//; +const RELEASE_RE = /^\d+\.\d+\.\d+$/; +const DATE_RE = /^\d{4}-\d{2}-\d{2}$/; const DEFAULT_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); @@ -101,7 +114,15 @@ export function validateRegistry(registry, { root = null } = {}) { for (const f of REQUIRED_FIELDS) { if (!(f in c)) errors.push(`${where}: missing required field "${f}"`); } - for (const f of ["id", "claim", "component", "version", "source_commit", "status"]) { + for (const f of [ + "id", + "claim", + "component", + "version", + "source_commit", + "assessed_release", + "status", + ]) { if (f in c && (typeof c[f] !== "string" || !c[f].trim())) { errors.push(`${where}: "${f}" must be a non-empty string`); } @@ -122,6 +143,31 @@ export function validateRegistry(registry, { root = null } = {}) { errors.push(`${where}: source_commit must be 7-40 lowercase hex characters`); } } + if ( + typeof c.assessed_release === "string" && + c.assessed_release && + c.assessed_release !== "unreleased" && + !RELEASE_RE.test(c.assessed_release) + ) { + errors.push(`${where}: assessed_release must be a release version (X.Y.Z) or "unreleased"`); + } + if ("counterexamples" in c) { + const ok = + Array.isArray(c.counterexamples) && + c.counterexamples.every( + (x) => + x && + typeof x === "object" && + DATE_RE.test(String(x.review ?? "")) && + typeof x.id === "string" && + x.id.trim() && + RESOLUTIONS.includes(x.resolution), + ); + if (!ok) + errors.push( + `${where}: counterexamples must be an array of {review: YYYY-MM-DD, id, resolution: ${RESOLUTIONS.join("|")}}`, + ); + } if ("evidence" in c) { if (!Array.isArray(c.evidence) || c.evidence.length === 0) { errors.push(`${where}: evidence must be a non-empty array of paths or URLs`); @@ -144,6 +190,84 @@ export function validateRegistry(registry, { root = null } = {}) { return errors; } +/** + * Cross-check each claim's `assessed_release` against the checkout's release tags: a claim + * marked "unreleased" whose source commit has since shipped is stale (record the release), and + * a claim naming a release must name one whose tag exists and contains its source commit. + * Returns null when the check cannot run — no git, or no `v*` tags (a shallow CI clone) — and + * skips a claim whose commit is not in this clone. + * @param {string} root + * @param {any} registry a valid registry + * @returns {string[]|null} + */ +export function releaseProblems(root, registry) { + const g = (args) => + execFileSync("git", args, { + cwd: root, + encoding: "utf8", + stdio: ["ignore", "pipe", "ignore"], + }).trim(); + const ok = (args) => { + try { + g(args); + return true; + } catch { + return false; + } + }; + let tags; + try { + tags = new Set(g(["tag", "--list", "v*"]).split("\n").filter(Boolean)); + } catch { + return null; + } + if (!tags.size) return null; + const firstRelease = new Map(); + const shippedIn = (commit) => { + if (!firstRelease.has(commit)) { + let first = null; + try { + first = + g(["tag", "--contains", commit, "--list", "v*", "--sort=v:refname"]) + .split("\n") + .filter(Boolean)[0] ?? null; + } catch {} + firstRelease.set(commit, first); + } + return firstRelease.get(commit); + }; + const problems = []; + for (const c of registry.claims) { + if (!ok(["cat-file", "-e", `${c.source_commit}^{commit}`])) continue; + const at = c.source_commit.slice(0, 12); + if (c.assessed_release === "unreleased") { + const first = shippedIn(c.source_commit); + if (first) + problems.push( + `${c.id}: assessed_release is "unreleased", but its source commit ${at} shipped in ${first} — record the release`, + ); + } else if (!tags.has(`v${c.assessed_release}`)) { + problems.push( + `${c.id}: assessed_release ${c.assessed_release} has no v${c.assessed_release} tag`, + ); + } else if (!ok(["merge-base", "--is-ancestor", c.source_commit, `v${c.assessed_release}`])) { + problems.push( + `${c.id}: release ${c.assessed_release} does not contain its source commit ${at}`, + ); + } + } + return problems; +} + +/** + * Stamp every claim assessed on unreleased code with the release that ships it — what + * `scripts/bump.mjs` does when it cuts `version`. Text-level, so the file's layout survives. + * @param {string} text claims.json + * @param {string} version + */ +export const stampRelease = (text, version) => + text.replace(/("assessed_release":\s*)"unreleased"/g, `$1"${version}"`); + /** Escape a value for a single Markdown table cell: backslashes first, then the pipes a * cell cannot contain (escaping only the pipes would let a trailing `\` undo the escape). */ export const cell = (s) => @@ -180,12 +304,17 @@ export function renderTable(registry) { `${claims.length} claims — ${counts.map(([s, n]) => `${s} ${n}`).join(" · ")}.`, `Assessed against commit${commits.length > 1 ? "s" : ""} ${commits.map((c) => `\`${c.slice(0, 12)}\``).join(", ")}${registry.as_of ? ` (as of ${registry.as_of})` : ""}.`, "", - "| ID | Status | Claim | Component · version | Evidence | Notes |", - "| --- | --- | --- | --- | --- | --- |", + "| ID | Status | Assessed | Claim | Component · version | Evidence | Notes |", + "| --- | --- | --- | --- | --- | --- | --- |", ]; for (const c of claims) { + // Review counterexamples are linked, not buried in prose: `2026-09-27 N01 fixed; …`. + const tested = (c.counterexamples ?? []).map((x) => `${x.review} ${x.id} ${x.resolution}`); + const notes = tested.length + ? `Review counterexamples: ${tested.join("; ")}. ${c.notes}` + : c.notes; lines.push( - `| \`${cell(c.id)}\` | **${cell(c.status)}** | ${cell(c.claim)} | ${cell(c.component)} · ${cell(c.version)} | ${c.evidence.map(evidenceLink).join("
")} | ${cell(c.notes)} |`, + `| \`${cell(c.id)}\` | **${cell(c.status)}** | ${cell(c.assessed_release)} · \`${cell(c.source_commit.slice(0, 8))}\` | ${cell(c.claim)} | ${cell(c.component)} · ${cell(c.version)} | ${c.evidence.map(evidenceLink).join("
")} | ${cell(notes)} |`, ); } return lines.join("\n"); @@ -292,6 +421,12 @@ export function run(argv, io = {}) { for (const p of problems) error(`registry: ${p}`); return 1; } + const releases = releaseProblems(root, registry); + if (releases === null) log("release check skipped: no v* tags in this checkout"); + else if (releases.length) { + for (const p of releases) error(`registry: ${p}`); + return 1; + } let failed = false; const readmeFile = path.join(root, README_PATH); diff --git a/src/atlas.js b/src/atlas.js index 1a1db42f..b334c773 100644 --- a/src/atlas.js +++ b/src/atlas.js @@ -20,8 +20,9 @@ import { contentHash, IGNORE_DIRS, toPosix } from "./util.js"; // Bumped whenever extraction or resolution changes shape: an atlas.json or per-file cache // from an older version is rebuilt, never trusted (v2 stored unresolved import specifiers; -// v3 filed every tsconfig path-alias import as an external package). -export const ATLAS_VERSION = 4; +// v3 filed every tsconfig path-alias import as an external package; v4 symbols carried no +// definition extent, `endLine`). +export const ATLAS_VERSION = 5; const JS_RULES = [ { @@ -661,16 +662,17 @@ function extractFile(path, root, preRead) { file: rel, line, }; - symbols.push({ + const sym = { name, kind, file: rel, line, id: node.id, qname: node.qname, - }); + }; + symbols.push(sym); nodes.push(node); - defs.push({ node, pos, kind }); + defs.push({ node, sym, pos, kind }); edges.push({ source: mod.id, target: node.id, @@ -692,6 +694,18 @@ function extractFile(path, root, preRead) { const parent = at >= 0 ? scopeAt(at) : null; if (parent && parent.node !== d.node && parent.end >= d.pos) d.node.local = true; } + // Definition EXTENTS (review N04/N08): the last line each definition spans, when its scope + // is known. Context delivery needs it to tell a whole definition from its first lines, and + // a dependency contract needs it to read a declaration whole. A JS declaration with no + // brace body (an overload, `declare function`, a `type` alias) ends with its statement. + // Locality is mirrored onto the symbol: a nested helper is never a cross-file dependency. + const endOf = new Map(scopes.map((sc) => [sc.node, sc.end])); + for (const d of defs) { + let end = endOf.get(d.node); + if (end === undefined && lex === "js") end = statementEnd(code, d.pos + d.node.name.length); + if (end !== undefined) d.node.endLine = d.sym.endLine = lines.at(Math.max(d.pos, end)); + if (d.node.local) d.sym.local = true; + } // Inheritance edges — `class X extends Y` (JS/TS) and `class X(Base, …)` (Python). Without // these the `inherits` edge weight was dead and a base-class change never appeared in blast diff --git a/src/cli/memory.js b/src/cli/memory.js index 30c1849f..48666baf 100644 --- a/src/cli/memory.js +++ b/src/cli/memory.js @@ -215,9 +215,11 @@ HANDLERS.ledger = async (argv) => { rt.learned ? ` retention: idle cut-off ${rt.cutoff} d = the longest idle stretch any claim came back from (${rt.comebacks} comebacks, typical gap ${rt.typicalGap} d; usage log spans ${rt.usageSpan} d)` : ` retention: not learned — ${rt.reason}`, - d?.boundary != null - ? ` duplicates: boundary ${d.boundary.toFixed(2)} (two components beat one: BIC ${d.bic2?.toFixed(1)} < ${d.bic1?.toFixed(1)}) · ${d.groups.length} group(s)` - : ` duplicates: none — ${d?.compared ? `one component fits the ${d.compared} nearest-neighbour similarities better` : "fewer than two claims of one kind are still live to compare"}`, + ` duplicates: ${d?.groups.length ?? 0} exact group(s) · ${ + d?.boundary != null + ? `near-duplicate boundary ${d.boundary.toFixed(2)} (two components beat one: BIC ${d.bic2?.toFixed(1)} < ${d.bic1?.toFixed(1)})` + : `no near-duplicate boundary — ${d?.compared ? `one component fits the ${d.compared} nearest-neighbour similarities better` : "fewer than two claims of one kind are still live to compare"}` + }`, "", ` archive: ${r.archive.length}`, ]; @@ -233,6 +235,19 @@ HANDLERS.ledger = async (argv) => { for (const c of conflicts.slice(0, 10)) lines.push(` ${c.a.slice(0, 12)} ↔ ${c.b.slice(0, 12)} ${c.conflicts}`); } + // Near-duplicates are never archived (review N02): similarity cannot see a swapped role, + // direction or number, or a detail one of them adds. A person merges them. + const proposed = d?.proposed ?? []; + if (proposed.length) { + lines.push( + "", + ` kept both — near-duplicates (retract one if they say the same thing): ${proposed.length}`, + ); + for (const p of proposed.slice(0, 10)) + lines.push( + ` ${p.a.slice(0, 12)} ↔ ${p.b.slice(0, 12)} similarity ${p.similarity.toFixed(2)}`, + ); + } lines.push( "", dryRun diff --git a/src/cli/routing.js b/src/cli/routing.js index 02d92004..b7c57c67 100644 --- a/src/cli/routing.js +++ b/src/cli/routing.js @@ -320,6 +320,7 @@ async function routeUniversalCli(argv) { ? ` attempt ${row.attemptId} was already recorded — not counted twice` : ` recorded ${row.model} ${row.passed ? "pass" : "fail"} (${row.provenance}) for task ${row.task} (.forge/route_outcomes.jsonl)`, ); + if (row.provenanceNote) console.log(` ! ${row.provenanceNote}`); } catch (e) { console.error(` ${e.message}`); process.exitCode = 1; @@ -349,7 +350,7 @@ async function routeUniversalCli(argv) { const fb = rec.fallback; if (fb) console.error( - ` fallback (does NOT meet the objective): ${fb.cascade.map((c) => c.model).join(" → ")} · P(success) ${fb.pSuccess.toFixed(2)} · expected $${fb.expectedCost.toFixed(3)} (up to $${fb.maxPossibleCost.toFixed(3)} if every attempt runs)`, + ` fallback (does NOT meet the objective): ${fb.cascade.map((c) => c.model).join(" → ")} · P(success) ${fb.pSuccess.toFixed(2)} · expected $${fb.expectedCost.toFixed(3)} (an estimated $${fb.estimatedCostIfAllAttemptsRun.toFixed(3)} if every attempt runs)`, ); process.exitCode = 1; return; @@ -368,7 +369,7 @@ async function routeUniversalCli(argv) { `\n ${paint("advice only", "warn")}: ${rec.unmapped.join(", ")} ${rec.unmapped.length === 1 ? "has" : "have"} no provider id — add one under "providers" in .forge/models.json, or pass --provider to route among models you can call (\`${BRAND.cli} route models\` lists who serves what)`, ); console.log( - `\n P(success) ${rec.pSuccess.toFixed(2)} · expected cost $${rec.expectedCost.toFixed(3)} (not a cap; up to $${rec.maxPossibleCost.toFixed(3)} if every attempt runs) · best single: ${rec.bestSingle.model} ${rec.bestSingle.pSuccess.toFixed(2)} at $${rec.bestSingle.expectedCost.toFixed(3)}`, + `\n P(success) ${rec.pSuccess.toFixed(2)} · expected cost $${rec.expectedCost.toFixed(3)} (not a cap; an estimated $${rec.estimatedCostIfAllAttemptsRun.toFixed(3)} if every attempt runs) · best single: ${rec.bestSingle.model} ${rec.bestSingle.pSuccess.toFixed(2)} at $${rec.bestSingle.expectedCost.toFixed(3)}`, ); console.log( ` ${rec.candidates} candidate model(s), ${rec.cascadesEvaluated} cascade(s) compared · fit: ${rec.fit.origin}`, diff --git a/src/cli/verification.js b/src/cli/verification.js index d5b8a490..56d9772e 100644 --- a/src/cli/verification.js +++ b/src/cli/verification.js @@ -112,8 +112,25 @@ HANDLERS.verify = async (argv) => { cov.uncovered.length ? ` — no verdict for ${cov.uncovered.join(", ")}` : "" }${cov.excluded.length ? ` (${cov.excluded.length} excluded)` : ""}`, ); + // Coverage strength (review N03): a verdict read from a package's own run is not the same + // evidence as one inferred from a recognized recursive root command, or declared by config. + const bases = Object.values(cov?.basis ?? {}); + if (bases.some((b) => b !== "measured")) { + const n = (b) => bases.filter((x) => x === b).length; + const how = [`${n("measured")} measured`]; + if (n("inferred")) + how.push(`${n("inferred")} inferred from \`${cov.rootRun?.command ?? "the root command"}\``); + if (n("declared")) how.push(`${n("declared")} declared (${cov.declared ?? "config"})`); + console.log(` coverage basis: ${how.join(" · ")}`); + } if (t.mutated) console.log(" ! the code changed while the tests ran — the verdict is not bound to it"); + // Scope of the fingerprint (review N05): what it could not bind, and what the repo declared + // outside the verified code. + if (t.unbound?.length) console.log(` ! not bound by the fingerprint: ${t.unbound.join(", ")}`); + const external = r.provenance.event?.external ?? []; + if (external.length) + console.log(` outside the proof: ${external.join(", ")} (declared in verify.external)`); console.log(` symbols checked: ${r.provenance.symbolsChecked}`); if (r.unknown.length) console.log( diff --git a/src/context.js b/src/context.js index 3aec0849..87f88301 100644 --- a/src/context.js +++ b/src/context.js @@ -115,26 +115,50 @@ export function requiredSet(root, task, { atlas = null, claims = [], nowDay = 0 // An item is one injectable unit with a COMPRESSION LADDER: granularity variants from // full text down to a one-line pointer. The optimizer may downgrade an item instead of // dropping it — compression is a lossy move with a known cost, chosen explicitly, never by -// scroll-off (spec §2). EVERY VARIANT CARRIES ITS OWN COVERAGE (review F03): the full file -// covers all its keys; a symbol span covers the definitions whose declaration line it shows; -// the first-25-lines head covers only what lies inside it; a pointer (`- read `) -// covers NOTHING — it creates a pending read obligation. Availability is not delivery. +// scroll-off (spec §2). EVERY VARIANT CARRIES ITS OWN COVERAGE (review F03, then N04): the +// full file covers all its keys; a span or the first-25-lines head covers a definition only +// when it shows the WHOLE definition, declaration through its last line (the atlas records +// each definition's extent, `endLine`). One that shows the declaration but cuts the body is +// PARTIAL — named in `partial` and still a pending read, so a 104-line function delivered as +// its first 41 lines is never reported as a delivered definition. A definition whose extent +// is unknown is covered by the whole file only. A pointer (`- read `) covers NOTHING — +// it creates a pending read obligation. Availability is not delivery. /** One variant: its rendered text, estimated tokens, the keys it satisfies, the keys it only - * points at (pending reads), and what it truncated. */ -const variant = (gran, text, covers, pending = [], truncated = null) => ({ + * points at (pending reads), what it truncated, and the definitions it shows only in part + * (`{key, shown, definition}` line ranges — a partial key is also pending). */ +const variant = (gran, text, covers, pending = [], truncated = null, partial = []) => ({ gran, text, tokens: tokensOf(text), covers, pending, ...(truncated ? { truncated } : {}), + ...(partial.length ? { partial } : {}), }); +/** Split definition needs by what a [from, to] line window shows of them: `covers` — the whole + * definition; `partial` — the declaration line but not its end (or an unknown end). */ +function windowCoverage(defs, from, to) { + const covers = []; + const partial = []; + for (const d of defs) { + if (d.line < from || d.line > to) continue; + if (Number.isFinite(d.endLine) && d.endLine <= to) covers.push(d.key); + else + partial.push({ + key: d.key, + shown: [from, to], + definition: [d.line, Number.isFinite(d.endLine) ? d.endLine : null], + }); + } + return { covers, partial }; +} + /** * All variants of one file, given the required keys it serves: `needs` entries are - * {key, kind: "def"|"file"|"tests", line?}. Ordered largest → smallest; a variant that is not - * smaller than the previous one is skipped. + * {key, kind: "def"|"file"|"tests", line?, endLine?}. Ordered largest → smallest; a variant + * that is not smaller than the previous one is skipped. */ function fileItem(root, rel, { needs, source, score }) { const text = readRel(root, rel); @@ -144,29 +168,32 @@ function fileItem(root, rel, { needs, source, score }) { const keys = needs.map((n) => n.key); const defs = needs.filter((n) => n.kind === "def" && Number.isFinite(n.line)); const variants = [variant("full", `// ${rel}\n${text}`, keys)]; - // Symbol span: the lines around the requested definitions, when the atlas knows them. + const windowed = (gran, label, from, to) => { + const { covers, partial } = windowCoverage(defs, from, to); + return variant( + gran, + `// ${rel}:${from}-${to} of ${total} (${label})\n${lines.slice(from - 1, to).join("\n")}`, + covers, + keys.filter((k) => !covers.includes(k)), + { shownLines: [from, to], totalLines: total }, + partial, + ); + }; if (defs.length) { const from = Math.max(1, Math.min(...defs.map((d) => d.line)) - SPAN_BEFORE); - const to = Math.min(total, Math.max(...defs.map((d) => d.line)) + SPAN_AFTER); - if (from > 1 || to < total) { - const covers = defs.filter((d) => d.line >= from && d.line <= to).map((d) => d.key); - variants.push( - variant( - "span", - `// ${rel}:${from}-${to} of ${total} (definition span)\n${lines.slice(from - 1, to).join("\n")}`, - covers, - keys.filter((k) => !covers.includes(k)), - { shownLines: [from, to], totalLines: total }, - ), - ); + // Whole definitions, when the atlas knows where every requested one ends. + if (defs.every((d) => Number.isFinite(d.endLine))) { + const to = Math.min(total, Math.max(...defs.map((d) => /** @type {number} */ (d.endLine)))); + if (from > 1 || to < total) variants.push(windowed("span", "whole definitions", from, to)); } + // The declaration and the start of the body: a smaller rung, partial for long bodies. + const to = Math.min(total, Math.max(...defs.map((d) => d.line)) + SPAN_AFTER); + if (from > 1 || to < total) variants.push(windowed("span", "definition span", from, to)); } if (total > HEAD_LINES) { - // The head covers a definition only if its declaration line is inside the head; a whole - // file or test file is never "covered" by its first 25 lines. - const covers = needs - .filter((n) => n.kind === "def" && Number.isFinite(n.line) && n.line <= HEAD_LINES) - .map((n) => n.key); + // The head covers a definition only if the WHOLE definition is inside it; a whole file or + // test file is never "covered" by its first 25 lines. + const { covers, partial } = windowCoverage(defs, 1, HEAD_LINES); variants.push( variant( "head", @@ -174,13 +201,18 @@ function fileItem(root, rel, { needs, source, score }) { covers, keys.filter((k) => !covers.includes(k)), { shownLines: [1, HEAD_LINES], totalLines: total }, + partial, ), ); } variants.push(variant("pointer", `- read ${rel}`, [], keys)); - const ladder = []; - for (const v of variants.sort((a, b) => b.tokens - a.tokens)) - if (!ladder.length || v.tokens < ladder[ladder.length - 1].tokens) ladder.push(v); + // The whole file is always the top rung; every lower rung is strictly smaller (a window + // whose label outweighs the lines it saves is no compression at all, and would otherwise + // outrank the full file while delivering less of it). + const [full, ...smaller] = variants; + const ladder = [full]; + for (const v of smaller.sort((a, b) => b.tokens - a.tokens)) + if (v.tokens < ladder[ladder.length - 1].tokens) ladder.push(v); return { id: `${source}:${rel}`, source, covers: keys, score, variants: ladder }; } @@ -188,16 +220,20 @@ function fileItem(root, rel, { needs, source, score }) { * Assemble the context for a task: pinned required items (downgraded before dropped), * optional items greedily by value density, and the missing set as derived questions. * - * Honesty contract (review F02/F03): + * Honesty contract (review F02/F03/N04): * - `tokens` is measured on the RENDERED block (labels and separators included) with the * chars/3.6 estimate (`tokenEstimate` says so). The block never exceeds `budget` by that * measure: when even pointers cannot fit, required items are DROPPED (lowest score first) * and reported, with `overflow: true`. - * - `covered` holds only keys whose content was actually delivered; `pending` holds keys the - * block merely points at (a pointer, or a partial span/head) — read obligations; `missing` - * holds keys neither delivered nor pointed at (unresolvable, or dropped on overflow). - * - `ok` means every required key was DELIVERED within budget: no missing, no pending, no - * overflow. It is syntactic delivery, not semantic sufficiency. + * - `covered` holds only keys whose content was actually delivered — for a definition, the + * WHOLE definition, declaration through its last line; `pending` holds keys the block + * merely points at (a pointer, a head or span that cuts the body) — read obligations; + * `partial` names the definitions whose declaration was shown but whose body was cut + * (shown lines vs the definition's extent); `missing` holds keys neither delivered nor + * pointed at (unresolvable, or dropped on overflow). + * - `ok` means every required key was DELIVERED within budget: no missing, no pending (so + * no partial definition), no overflow. It is syntactic delivery, not semantic + * sufficiency. * - Optional items are chosen greedily by value density (score per token) with per-source * diminishing returns — a heuristic, with no knapsack or set-cover guarantee (the * per-source discount breaks the preconditions those guarantees need). @@ -217,7 +253,7 @@ export function assemble( // --- build candidate items, keyed by what they cover ------------------------------- // File-backed keys are grouped per file first, so one file is one item whose variants // know exactly which of its keys each one delivers. - /** @type {Map} */ + /** @type {Map} */ const files = new Map(); const need = (rel, n, source, score) => { const f = files.get(rel) ?? { needs: [], source, score }; @@ -235,7 +271,8 @@ export function assemble( if (!r.resolvable) continue; if (r.kind === "def") { const hit = atlasQuery(atlas, r.name).find((s) => s.name === r.name || s.qname === r.name); - if (hit?.file) need(hit.file, { key: r.key, kind: "def", line: hit.line }, "def", 1); + if (hit?.file) + need(hit.file, { key: r.key, kind: "def", line: hit.line, endLine: hit.endLine }, "def", 1); } else if (r.kind === "file") { need(r.name, { key: r.key, kind: "file" }, "def", 1); } else if (r.kind === "tests") { @@ -366,6 +403,12 @@ export function assemble( const missing = required.filter( (r) => !r.resolvable || (!covered.has(r.key) && !pending.has(r.key)), ); + // Definitions shown only in part (declaration delivered, body cut — review N04). Each is + // also pending: the rest of the body is a read obligation, never "delivered". + const partial = chosen + .flatMap((c) => c.item.variants[c.v].partial ?? []) + .filter((p) => !covered.has(p.key)) + .sort((a, b) => (a.key < b.key ? -1 : 1)); const questions = missing .filter((r) => !r.resolvable) .map((r) => @@ -388,6 +431,7 @@ export function assemble( required: required.map((r) => r.key), covered: [...covered].sort(), pending: [...pending].sort(), + partial, missing: missing.map((r) => r.key), questions, truncated, @@ -419,6 +463,10 @@ export function renderContext(r) { lines.push("", " pending reads (pointed at, not delivered — read before acting):"); for (const p of r.pending) lines.push(` - ${p}`); } + for (const p of r.partial ?? []) + lines.push( + ` ~ ${p.key}: only lines ${p.shown[0]}-${p.shown[1]} shown of a definition spanning ${p.definition[0]}-${p.definition[1] ?? "?"} — the body is not delivered`, + ); for (const t of r.truncated ?? []) if (t.omitted?.length) lines.push( diff --git a/src/learn_consolidate.js b/src/learn_consolidate.js index 82cc0ba5..9b5d1da8 100644 --- a/src/learn_consolidate.js +++ b/src/learn_consolidate.js @@ -4,20 +4,25 @@ // send every lesson to a model with "DROP anything … contradicted" and rewrite the store // from its answer: pruning memory by the model's own judgment, which the research rejects // (white paper §3, memory residual gap; §7.1, val = validity from an external oracle). -// Here nothing is judged, reworded or invented: -// - MERGE: an exact duplicate (normalized text) or a near-duplicate (MinHash Jaccard ≥ τ, -// the ledger's own consolidation threshold, ledger.clusters) within one project -// collapses into its first occurrence — but only when the semantic guard finds no -// behaviour-bearing difference (review F16): "Enable authentication…" and "Disable -// authentication…" overlap almost entirely and are OPPOSITE rules, so they are kept apart -// and reported as a conflict for a person to resolve. -// - DROP: only on ledger ground truth. A lesson is dropped when its best-matching ledger -// claim (lesson/fact, Jaccard ≥ τ against claimText, and not reversed by polarity, -// operators, numbers or literals) is dormant (ledger.isDormant: its oracle-evidenced val -// fell below DORMANT_VAL and no confirmation restored it) or retracted (tombstoned). An -// ARCHIVED claim is not a refuted one (review F15): the attic also holds claims archived for -// idleness or as duplicates, so an attic claim refutes only when its own logs say so -// (tombstone/dormant), and a deduplicated one defers to the claim that survived it. +// Here nothing is judged, reworded or invented — and similarity never deletes a rule: +// - MERGE: only an EXACT duplicate within one project collapses into its first occurrence: +// the same text up to whitespace and trailing sentence punctuation, with the semantic +// guard confirming that no literal or code layout depended on that difference (review +// N02). A near-duplicate (MinHash Jaccard ≥ τ) is NOT merged: "Allow admins and deny +// guests…" and "Deny admins and allow guests…" share every token, polarity words +// included, so no token-level check can tell them apart — relational reversals (who gets +// which action), swapped numbers (read timeout 5 s / write 30 s) and moved negation scope +// all look identical to it. Both lessons are KEPT and the pair is reported as a +// `proposed` grouping for a person to merge by hand; a pair whose tokens do conflict +// ("Enable…" / "Disable…", review F16) is reported as a conflict instead. +// - DROP: only on ledger ground truth, and only for the SAME rule. A lesson is dropped when +// a ledger claim (lesson/fact) with exactly its text (same equality as MERGE) is dormant +// (ledger.isDormant: its oracle-evidenced val fell below DORMANT_VAL and no confirmation +// restored it) or retracted (tombstoned). A refuted claim that is merely SIMILAR does not +// drop anything — it may be the opposite rule — so the lesson is kept and `flagged` for +// review. An ARCHIVED claim is not a refuted one (review F15): the attic also holds claims +// archived for idleness or as duplicates, so an attic claim refutes only when its own logs +// say so (tombstone/dormant), and a deduplicated one defers to the claim that survived it. // A lesson with no matching claim is KEPT: absence of evidence is not refutation. // Claims are matched only within the lesson's project (a repo whose directory name is the // project), so a lesson refuted in one repo is not dropped from another. @@ -39,9 +44,9 @@ import { claimText, isDormant, jaccard, sketch } from "./ledger.js"; import { archiveRecord, getClaimByPrefix, loadClaims, repoLedger } from "./ledger_store.js"; import { describeConflicts, - FLIP_KINDS, - sameSemantics, + sameStatement, semanticConflicts, + statementKey, } from "./semantic_guard.js"; import { epochDay } from "./util.js"; @@ -54,13 +59,6 @@ export const GENERAL = "General"; * path read a different folder than the one the lessons were written to. */ export const learnedDir = () => join(process.env.HOME || homedir(), ".claude", "skills", "learned"); -const norm = (s) => - String(s) - .toLowerCase() - .replace(/[`*_"'.,;:!?()[\]{}]/g, " ") - .replace(/\s+/g, " ") - .trim(); - /** * Parse learned-lesson markdown into `{project, text}` entries. Understands both shapes: * the session-learner's `## 2026-07-04 12:00 — ` headers and a consolidated @@ -124,13 +122,18 @@ function refutation(claim, nowDay, byId = new Map(), depth = 0) { } /** - * Consolidate deterministically: merge duplicates, drop only ledger-refuted lessons. + * Consolidate deterministically: merge exact duplicates, drop only lessons whose exact ledger + * claim is refuted, and REPORT (never act on) similarity: `proposed` near-duplicates to merge + * by hand, `conflicts` for similar pairs that differ in a behaviour-bearing token, `flagged` + * lessons that resemble a refuted claim. * @param {Learned[]} entries * @param {{claims?: LedgerClaim[], nowDay?: number, tau?: number}} [opts] each claim may * carry `project` (the repo directory name); a claim without one matches any project. * @returns {{kept: Learned[], merged: {text: string, into: string}[], * dropped: {project: string, text: string, claim: string, reason: string}[], - * conflicts: {project: string, text: string, other: string, conflicts: string}[]}} + * conflicts: {project: string, text: string, other: string, conflicts: string}[], + * proposed: {project: string, text: string, similar: string, similarity: number}[], + * flagged: {project: string, text: string, claim: string, reason: string}[]}} */ export function consolidateLearned( entries, @@ -145,59 +148,69 @@ export function consolidateLearned( s: sketch(claimText(c)), why: refutation(c, nowDay, byId), })); - /** @type {(Learned & {s: any, n: string})[]} */ + /** @type {(Learned & {s: any})[]} */ const kept = []; const merged = []; const dropped = []; /** @type {{project: string, text: string, other: string, conflicts: string}[]} */ const conflicts = []; + /** @type {{project: string, text: string, similar: string, similarity: number}[]} */ + const proposed = []; + /** @type {{project: string, text: string, claim: string, reason: string}[]} */ + const flagged = []; + const claimRef = (c) => String(c.id ?? "").slice(0, 12); for (const e of entries) { const text = String(e.text || "").trim(); if (!text) continue; const project = e.project || GENERAL; const s = sketch(text); - const n = norm(text); - // Similar is not the same (F16): a close pair that differs in polarity, operators, - // numbers, literals, identifiers or paths is two rules — keep both, report the conflict. - let dup = null; - for (const k of kept) { - if (k.project !== project || (k.n !== n && jaccard(k.s, s) < tau)) continue; - const differs = semanticConflicts(k.text, text); - if (!differs.length) { - dup = k; - break; - } - conflicts.push({ project, text, other: k.text, conflicts: describeConflicts(differs) }); - } + const dup = kept.find((k) => k.project === project && sameStatement(k.text, text)); if (dup) { merged.push({ text, into: dup.text }); continue; } - let best = null; + // Similar is not the same (F16/N02): a close pair is reported, never merged — as a + // conflict when a behaviour-bearing token differs, else as a proposal to review. + for (const k of kept) { + if (k.project !== project) continue; + const j = jaccard(k.s, s); + if (j < tau && statementKey(k.text) !== statementKey(text)) continue; + const differs = semanticConflicts(k.text, text); + if (differs.length) + conflicts.push({ project, text, other: k.text, conflicts: describeConflicts(differs) }); + else proposed.push({ project, text, similar: k.text, similarity: Number(j.toFixed(2)) }); + } + let exact = null; + let near = null; for (const u of usable) { if (u.c.project && project !== GENERAL && u.c.project !== project) continue; + if (sameStatement(u.text, text)) { + if (!exact || (u.why && !exact.why)) exact = u; + continue; + } const j = jaccard(u.s, s); - // A claim saying the OPPOSITE (a flipped polarity/operator/number/literal) is not this - // lesson's evidence, however similar the words. - if (j >= tau && (!best || j > best.j) && sameSemantics(u.text, text, { kinds: FLIP_KINDS })) - best = { ...u, j }; + if (u.why && j >= tau && (!near || j > near.j)) near = { ...u, j }; } - if (best?.why) { - dropped.push({ + if (exact?.why) { + dropped.push({ project, text, claim: claimRef(exact.c), reason: exact.why }); + continue; + } + if (!exact && near) + flagged.push({ project, text, - claim: String(best.c.id ?? "").slice(0, 12), - reason: best.why, + claim: claimRef(near.c), + reason: `similar (${near.j.toFixed(2)}) to a claim ${near.why} — kept: it may be the opposite rule`, }); - continue; - } - kept.push({ project, text, s, n }); + kept.push({ project, text, s }); } return { kept: kept.map(({ project, text }) => ({ project, text })), merged, dropped, conflicts, + proposed, + flagged, }; } @@ -271,7 +284,7 @@ export function consolidateDir({ ...(existsSync(join(dir, "CONSOLIDATED.md")) ? ["CONSOLIDATED.md"] : []), ...monthly, ]; - const none = { kept: [], merged: [], dropped: [], conflicts: [] }; + const none = { kept: [], merged: [], dropped: [], conflicts: [], proposed: [], flagged: [] }; if (!inputs.length) return { ok: true, row: "nothing", dir, inputs, ...none }; const entries = inputs.flatMap((f) => parseLearned(readFileSync(join(dir, f), "utf8"))); if (!entries.length) return { ok: true, row: "empty", dir, inputs, ...none }; @@ -309,6 +322,22 @@ export function renderReport(r) { ` ! [${c.project}] ${c.text.slice(0, 60)} ↔ ${c.other.slice(0, 60)} — ${c.conflicts}`, ); } + const proposed = r.proposed ?? []; + if (proposed.length) { + lines.push( + ` kept both — similar, merge by hand if they are the same rule: ${proposed.length}`, + ); + for (const p of proposed.slice(0, 20)) + lines.push( + ` ~ [${p.project}] ${p.text.slice(0, 60)} ↔ ${p.similar.slice(0, 60)} (${p.similarity})`, + ); + } + const flagged = r.flagged ?? []; + if (flagged.length) { + lines.push(` kept — resemble a refuted ledger claim (review): ${flagged.length}`); + for (const f of flagged.slice(0, 20)) + lines.push(` ? [${f.project}] ${f.text.slice(0, 80)} — ${f.reason} (claim ${f.claim})`); + } return lines.join("\n"); } diff --git a/src/ledger.js b/src/ledger.js index 6247185a..0973ab55 100644 --- a/src/ledger.js +++ b/src/ledger.js @@ -95,6 +95,10 @@ export const DORMANT_VAL = 0.35; * - Non-string values: numbers, booleans and null serialize as JSON.stringify does. */ const canonText = (s) => s.normalize("NFC").replace(/\r\n/g, "\n"); +/** Whether the ledger stores this string byte for byte — canonicalization is the identity on + * it. Content that must survive storage EXACTLY (an inline code artifact, review N01) is + * checked with this rather than silently NFC/LF-folded on write. @param {string} s */ +export const storesVerbatim = (s) => canonText(String(s)) === String(s); // The rule BEFORE the CRLF fold. A claim minted on a CRLF checkout carries the old address // in its FILENAME, and recomputing it under the new rule made the file fail its own address // check: the claim did not degrade, it vanished — loadClaims returned nothing for it. The diff --git a/src/ledger_retention.js b/src/ledger_retention.js index baefa2c1..28615081 100644 --- a/src/ledger_retention.js +++ b/src/ledger_retention.js @@ -18,16 +18,20 @@ // F1 was tried first: it archived a claim used every 3 days on the day it fell due.) // The rule only switches on once the usage log spans longer than that gap. Before // then, "not used again" only means "use was not recorded". -// - GROUP near-duplicates of one kind: each claim's nearest-neighbour similarity is -// modelled as one Gaussian or two (hard split, Otsu), BIC picks the model, and only a -// two-component fit yields a duplicate boundary (where the two posteriors are equal). +// - GROUP exact duplicates of one kind — the same statement up to whitespace and trailing +// punctuation (semantic_guard.sameStatement) — and archive all but one. NEAR-duplicates +// are only PROPOSED (review N02): each claim's nearest-neighbour similarity is modelled as +// one Gaussian or two (hard split, Otsu), BIC picks the model, and a two-component fit +// yields a boundary (where the two posteriors are equal) above which pairs are reported for +// a person to merge — never archived, because similarity cannot see a relational swap +// ("allow admins, deny guests" / "deny admins, allow guests") or a detail one of them adds. // A ledger without duplicates yields none. // // Everything this plans is reversible: an archived claim keeps its bytes in the attic and // its logs in place, and any new evidence brings it back (ledger_store appendRecord). import { claimText, isDormant, jaccard, SKETCH_K, sketch, val } from "./ledger.js"; -import { describeConflicts, semanticConflicts } from "./semantic_guard.js"; +import { describeConflicts, sameStatement, semanticConflicts } from "./semantic_guard.js"; /** @typedef {{id: string, kind?: string, body?: any, provenance?: {t?: number}, * evidence?: {t?: number}[], tombstone?: {t?: number} | null}} Claim */ @@ -174,17 +178,20 @@ export function similarityBoundary(xs, k = SKETCH_K) { } /** - * Near-duplicate groups among live claims of the same kind, with one survivor each: the - * highest val, then the most evidence, then the earliest minted, then the smallest id. - * Similarity only PROPOSES a duplicate (review F16): a close pair whose texts differ in - * polarity, operators, numbers, literals, identifiers or paths ("Enable authentication…" vs - * "Disable authentication…") is never grouped — it is reported in `conflicts`, both claims - * stay live, and a person decides. + * Duplicate groups among live claims of the same kind, with one survivor each: the highest + * val, then the most evidence, then the earliest minted, then the smallest id. Only EXACT + * duplicates are grouped (review N02: the same statement, semantic_guard.sameStatement) — + * whatever the learned boundary, since the same statement is always a duplicate. A pair at or + * above the learned boundary that is not the same statement is only REPORTED: in `conflicts` + * when it differs in polarity, operators, numbers, literals, identifiers, paths or code layout + * ("Enable authentication…" vs "Disable authentication…", review F16), else in `proposed` — a + * near-duplicate a person may merge. Both stay live either way. * @param {Claim[]} claims live (servable) claims * @param {number} nowDay * @returns {{boundary: number | null, bic1?: number, bic2?: number | null, compared: number, * groups: {keep: string, drop: {id: string, similarity: number}[]}[], - * conflicts: {a: string, b: string, similarity: number, conflicts: string}[]}} + * conflicts: {a: string, b: string, similarity: number, conflicts: string}[], + * proposed: {a: string, b: string, similarity: number}[]}} */ export function duplicateGroups(claims, nowDay) { const byKind = new Map(); @@ -208,18 +215,9 @@ export function duplicateGroups(claims, nowDay) { } nn.push(...best); } - if (!nn.length) return { boundary: null, compared: 0, groups: [], conflicts: [] }; + if (!nn.length) return { boundary: null, compared: 0, groups: [], conflicts: [], proposed: [] }; const fit = similarityBoundary(nn); - if (fit.boundary == null) - return { - boundary: null, - bic1: fit.bic1, - bic2: fit.bic2, - compared: nn.length, - groups: [], - conflicts: [], - }; - // Union-find over the pairs at or above the boundary. + // Union-find over the EXACT-duplicate pairs. const parent = new Map(); const find = (x) => { while (parent.get(x) !== x) { @@ -229,25 +227,31 @@ export function duplicateGroups(claims, nowDay) { return x; }; const pairKey = (a, b) => (a < b ? `${a}\n${b}` : `${b}\n${a}`); - /** @type {Map} similarity of each pair at or above the boundary */ + /** @type {Map} similarity of each exact-duplicate pair */ const close = new Map(); /** @type {{a: string, b: string, similarity: number, conflicts: string}[]} */ const conflicts = []; + /** @type {{a: string, b: string, similarity: number}[]} */ + const proposed = []; for (const p of pairs) { - if (p.sim < fit.boundary) continue; - const differs = semanticConflicts(statementOf(p.i.c), statementOf(p.j.c)); - if (differs.length) { + const a = statementOf(p.i.c); + const b = statementOf(p.j.c); + if (sameStatement(a, b)) { + for (const it of [p.i, p.j]) if (!parent.has(it.c.id)) parent.set(it.c.id, it.c.id); + parent.set(find(p.i.c.id), find(p.j.c.id)); + close.set(pairKey(p.i.c.id, p.j.c.id), p.sim); + continue; + } + if (fit.boundary == null || p.sim < fit.boundary) continue; + const differs = semanticConflicts(a, b); + if (differs.length) conflicts.push({ a: p.i.c.id, b: p.j.c.id, similarity: p.sim, conflicts: describeConflicts(differs), }); - continue; - } - for (const it of [p.i, p.j]) if (!parent.has(it.c.id)) parent.set(it.c.id, it.c.id); - parent.set(find(p.i.c.id), find(p.j.c.id)); - close.set(pairKey(p.i.c.id, p.j.c.id), p.sim); + else proposed.push({ a: p.i.c.id, b: p.j.c.id, similarity: p.sim }); } const members = new Map(); const claimById = new Map(claims.map((c) => [c.id, c])); @@ -283,13 +287,15 @@ export function duplicateGroups(claims, nowDay) { }) .filter((g) => g.drop.length) .sort((a, b) => (a.keep < b.keep ? -1 : 1)); + const byPair = (x, y) => (x.a < y.a ? -1 : x.a > y.a ? 1 : x.b < y.b ? -1 : 1); return { boundary: fit.boundary, bic1: fit.bic1, bic2: fit.bic2, compared: nn.length, groups, - conflicts: conflicts.sort((x, y) => (x.a < y.a ? -1 : x.a > y.a ? 1 : x.b < y.b ? -1 : 1)), + conflicts: conflicts.sort(byPair), + proposed: proposed.sort(byPair), }; } @@ -355,7 +361,7 @@ export function retentionPlan(claims, uses, nowDay, { halfLife, duplicates = fal for (const d of g.drop) archive.push({ id: d.id, - reason: `near-duplicate of ${g.keep.slice(0, 12)} (similarity ${d.similarity.toFixed(2)} ≥ learned ${dup.boundary?.toFixed(2)})`, + reason: `duplicate of ${g.keep.slice(0, 12)} (the same statement)`, cause: "duplicate", survivor: g.keep, }); diff --git a/src/reuse.js b/src/reuse.js index e31d64ff..822ba769 100644 --- a/src/reuse.js +++ b/src/reuse.js @@ -5,11 +5,21 @@ // never asserted) AND its dependencies still resolving in the atlas — stale or // discredited code silently stops being served, because the cache is pruned by // ground truth, not by an LRU. +import { createHash } from "node:crypto"; import { existsSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { has as atlasHas } from "./atlas.js"; import { claimSim, simLabel } from "./embed.js"; -import { isDormant, jaccard, mintClaim, outcomeRecord, SKETCH_K, sketch, val } from "./ledger.js"; +import { + isDormant, + jaccard, + mintClaim, + outcomeRecord, + SKETCH_K, + sketch, + storesVerbatim, + val, +} from "./ledger.js"; import { appendEvidence, loadClaims, putClaim, readEvidence, repoLedger } from "./ledger_store.js"; import { record as recordMetric } from "./metrics.js"; import { describeConflicts, semanticConflicts } from "./semantic_guard.js"; @@ -32,27 +42,34 @@ export const NEAR_COS = 0.85; export const ADAPT_COS = 0.7; // --------------------------------------------------------------------------- -// Normalization — TWO forms, because "the same neighbourhood" and "the same task" are -// different questions (review C9, then review F04): -// • specKey (IDENTITY) — LOSSLESS except for whitespace: Unicode NFC and runs of whitespace -// collapsed, NOTHING else. Case, operators, literals, paths and punctuation all stay: the -// old identity form lowercased and trimmed punctuation per token, so `accept ages >= 18` -// and `<= 18`, `return "ADMIN"` and `"admin"`, `set enabled = true` and `!= true`, and -// `getURL` and `getUrl` shared ONE exact key — an exact hit could serve the opposite -// requirement. Two specs now share an exact key only when they are the same text. +// Identity and normalization — the same neighbourhood, the same instruction and the same +// task are different questions (review C9, F04, then N01): +// • specDigest (EXACT IDENTITY) — NO normalization at all: a digest of the spec's code +// units. The v2 key still collapsed whitespace and folded Unicode NFC, so `return "a b"` +// and `return "a b"` shared one key and an exact hit served the two-space string for a +// one-space request (review N01). Whitespace inside a literal is data, indentation is +// structure (Python, YAML, a Makefile tab), a regex's spaces are its pattern, and composed +// and decomposed Unicode are different strings to a program — no rewrite, however +// "harmless", may sit at the as-is boundary. Stored as hex (`body.keyHash`), which the +// ledger's own NFC/LF canonicalization of claim text cannot fold. +// • specKey (IDENTITY TEXT) — the spec as given: what the near tier sketches (the sketch is +// whitespace- and case-insensitive by construction) and the semantic guard compares. // • normalizeSpec (SHAPE) — volatile literals and identifiers become typed placeholders, -// so those two specs still land in one near-NEIGHBOURHOOD and the adapt tier can offer -// the artifact as a verified starting point ("generate only the delta"). -// Similarity (the near tier's MinHash over the identity key, or an embedding cosine) finds +// so specs that differ only in those still land in one near-NEIGHBOURHOOD and the adapt +// tier can offer the artifact as a verified starting point ("generate only the delta"). +// Similarity (the near tier's MinHash over the identity text, or an embedding cosine) finds // NEIGHBOURS; it never by itself authorizes serving code as-is: a near candidate must also -// pass the semantic guard (same operators, numbers, literals, identifiers, paths and -// polarity words), else it is only offered at the adapt tier, for review. -// Both tokenizers are Unicode-aware: the ASCII `\w` trim erased every Arabic (or Chinese, -// or Greek) word, so any two non-ASCII specs normalized to "" and collided as exact. +// pass the semantic guard (same operators, numbers, literals, identifiers, paths, polarity +// words and code layout), else it is only offered at the adapt tier, for review. So a spec +// that differs from a verified one only in whitespace reaches near at best — and not even +// that when the whitespace sits in a literal, a code block or beside a code token. +// The tokenizers are Unicode-aware: the ASCII `\w` trim erased every Arabic (or Chinese, or +// Greek) word, so any two non-ASCII specs normalized to "" and collided as exact. // --------------------------------------------------------------------------- -/** The identity-key format version. v1 keys (pre-F04) were lossy, so they never exact-hit. */ -export const KEY_VERSION = 2; +/** The identity-key format version. v1 keys (pre-F04) were lossy; v2 (pre-N01) collapsed + * whitespace and folded NFC. Neither ever reaches the exact tier again. */ +export const KEY_VERSION = 3; const NUM_RE = /^-?\d[\d.,_]*$/; const PATH_RE = /[\\/]|\.(?:m?[jt]sx?|py|go|rs|java|rb|json|ya?ml|toml|md|css|html)$/i; @@ -67,10 +84,18 @@ const IDENT_RE = /^(?:[a-z][a-z0-9]*[A-Z]|[A-Z][a-z0-9]+[A-Z]|\w+_\w+|\w+\.\w+)\ // marks of any script, plus the code punctuation the classifiers below key on. const TRIM_RE = /^[^\p{L}\p{N}\p{M}_"'`./\\-]+|[^\p{L}\p{N}\p{M}_"'`./\\-]+$/gu; -/** Identity normalization: NFC + whitespace runs collapsed + ends trimmed — nothing else. The - * exact key: two specs match here only when they are the SAME text. */ +/** The identity TEXT of a spec: the spec exactly as given (review N01). */ export function specKey(text) { - return String(text).normalize("NFC").trim().split(/\s+/).filter(Boolean).join(" "); + return String(text ?? ""); +} + +/** The EXACT identity (review N01): sha256 over the spec's UTF-16 code units — lossless for + * every JS string, lone surrogates included, so two specs share it only when they are the + * same string. @param {unknown} text @returns {string} hex */ +export function specDigest(text) { + return createHash("sha256") + .update(Buffer.from(String(text ?? ""), "utf16le")) + .digest("hex"); } /** Deterministic, pure spec SHAPE normalization (unit-tested surface). */ @@ -90,15 +115,18 @@ export function normalizeSpec(text) { .join(" "); } -/** The cache keys: `exact` (identity key + graph-slice context), `keySketch` (what the near - * tier measures) and `sketch` (the shape form the adapt tier and the LSH prefilter use). */ +/** The cache keys: `digest` (the exact identity), `exact` (identity + graph-slice context), + * `keySketch` (what the near tier measures) and `sketch` (the shape form the adapt tier and + * the LSH prefilter use). */ export function fingerprint(spec, slice = "") { const norm = normalizeSpec(spec); const key = specKey(spec); + const digest = specDigest(spec); return { norm, key, - exact: contentHash(`${key}\u0000${slice}`), + digest, + exact: contentHash(`${digest}\u0000${slice}`), sketch: sketch(norm), keySketch: sketch(key), }; @@ -142,26 +170,41 @@ export function artifactClaim( iface = [], deps = [], depContracts = {}, + depSources = {}, code, lang = "", form = "function", }, t = 0, ) { + // The ledger stores claim text NFC- and LF-normalized. For inline CODE that is a silent + // rewrite of the verified bytes (a CRLF inside a string literal, a decomposed "é"), so it + // is refused, with the lossless alternative named (review N01). + if (typeof code?.inline === "string" && !storesVerbatim(code.inline)) + return { + ok: false, + reason: + "inline code has CRLF line endings or non-NFC Unicode, which the ledger's canonical storage would rewrite — mint it as a file pointer ({path, sha256}) so the verified bytes are the served bytes", + }; return mintClaim({ kind: "artifact", body: { code: code ?? {}, - // name → fingerprint of the dependency's declaration at mint time (review F05): a - // same-name dependency whose signature changed invalidates the artifact. + // name → fingerprint of the dependency's declaration at mint time (review F05/N08): a + // dependency whose contract changed invalidates the artifact; `null` = not established + // at mint (revalidation then says unknown, never valid). `depSources` pins the module + // each name was imported from, so a same-name symbol elsewhere is never compared. ...(Object.keys(depContracts).length ? { depContracts: sortedObject(depContracts) } : {}), + ...(Object.keys(depSources).length ? { depSources: sortedObject(depSources) } : {}), deps: [...deps].sort(), form, iface: [...iface].sort(), - // `key` is the LOSSLESS identity form (exact + near); `spec` stays the SHAPE form (adapt - // and the LSH prefilter). A pre-C9 artifact has no key and can only reach adapt; a - // pre-F04 key (no keyV) was lossy and never reaches exact. + // `keyHash` is the exact identity (review N01), `key` the identity text the near tier + // and the semantic guard read, `spec` the SHAPE form (adapt and the LSH prefilter). A + // pre-C9 artifact has no key and can only reach adapt; a key from an older version + // (keyV < 3) was lossy and never reaches exact. key: specKey(spec), + keyHash: specDigest(spec), keyV: KEY_VERSION, lang, slice, @@ -206,36 +249,182 @@ export function mintArtifact(dir, fields, { evidence, t = 0 } = {}) { const sortedObject = (o) => Object.fromEntries(Object.entries(o).sort(([a], [b]) => (a < b ? -1 : 1))); +// --------------------------------------------------------------------------- +// Dependency contracts (review F05, then N08). A caller relies on a dependency's DECLARATION — +// its name, parameters (destructured keys, defaults and nested patterns included), modifiers +// and annotations — so that is what is fingerprinted. The v1 scanner cut the declaration at +// its first `{` or `=>`, which for `function calc({a})` is the destructuring brace: both +// `calc({a})` and `calc({b})` hashed as `export function calc(`, and an artifact whose +// dependency changed its contract kept serving as valid. Now the declaration is read WHOLE +// (the atlas knows where each definition ends), lexed — comments dropped, whitespace between +// tokens ignored, string literals kept verbatim — and only the BODY is cut: the final +// top-level block of a function, what follows a top-level `=>`, what follows a Python `def`'s +// colon. A class, type or plain value has no body to cut: all of it is the contract. And the +// dependency is the definition the artifact's own import BINDS to (its module identity, from +// the atlas's structural import resolution), never whichever same-name symbol sorts first. +// Whatever cannot be established — an unknown extent, a name defined in several files with no +// import binding to choose one — is null: "unknown", so the hit requires revalidation. +// --------------------------------------------------------------------------- + +/** The contract format. Contracts recorded in another format are never compared: unknown. */ +export const CONTRACT_VERSION = 2; +const CONTRACT_PREFIX = `v${CONTRACT_VERSION}:`; + +/** Split source into contract tokens: identifiers/numbers, string literals (verbatim, their + * whitespace included), and punctuation, with `=>`, `->`, `...` kept whole. Comments and + * inter-token whitespace are dropped, so a reformat never changes a contract. Deterministic + * on any input — an odd construct (a regex, a Rust lifetime) still yields stable tokens. + * @param {string} text @param {"js"|"py"} lang */ +function contractTokens(text, lang) { + const toks = []; + const n = text.length; + const hashComments = lang === "py"; + let i = 0; + while (i < n) { + const c = text[i]; + if (/\s/.test(c)) i += 1; + else if (hashComments ? c === "#" : c === "/" && text[i + 1] === "/") { + const nl = text.indexOf("\n", i); + i = nl < 0 ? n : nl; + } else if (!hashComments && c === "/" && text[i + 1] === "*") { + const close = text.indexOf("*/", i + 2); + i = close < 0 ? n : close + 2; + } else if (c === '"' || c === "'" || c === "`") { + const triple = lang === "py" && text.startsWith(c.repeat(3), i); + let j = i + (triple ? 3 : 1); + if (triple) { + const close = text.indexOf(c.repeat(3), j); + j = close < 0 ? n : close + 3; + } else { + while (j < n && text[j] !== c) j += text[j] === "\\" ? 2 : 1; + j = Math.min(n, j + 1); + } + toks.push(text.slice(i, j)); + i = j; + } else if (/[\p{L}\p{N}_$]/u.test(c)) { + let j = i + 1; + while (j < n && /[\p{L}\p{N}_$]/u.test(text[j])) j += 1; + toks.push(text.slice(i, j)); + i = j; + } else { + const op = ["...", "=>", "->"].find((o) => text.startsWith(o, i)) ?? c; + toks.push(op); + i += op.length; + } + } + return toks; +} + +const OPEN = new Set(["(", "[", "{"]); +const CLOSE = new Set([")", "]", "}"]); + +/** Whether a `const` declaration's value is a function (arrow or function expression) — then + * its body is cut like a function's; any other value IS the contract. */ +function isFunctionValue(toks) { + const eq = toks.indexOf("="); + if (eq < 0) return false; + let k = eq + 1; + if (toks[k] === "async") k += 1; + if (toks[k] === "function") return true; + let depth = 0; + for (let i = k; i < toks.length; i++) { + if (OPEN.has(toks[i])) depth += 1; + else if (CLOSE.has(toks[i])) depth -= 1; + else if (toks[i] === "=>" && depth === 0) return true; + else if (depth === 0 && (toks[i] === ";" || toks[i] === ",")) return false; + } + return false; +} + +/** The contract part of one declaration's tokens: everything but a function's body. */ +function declarationHead(toks, kind, lang) { + if (kind === "class" || kind === "type") return toks; + if (lang === "py") { + let depth = 0; + let params = false; + for (let i = 0; i < toks.length; i++) { + if (OPEN.has(toks[i])) { + depth += 1; + if (toks[i] === "(") params = true; + } else if (CLOSE.has(toks[i])) depth -= 1; + else if (toks[i] === ":" && depth === 0 && params) return toks.slice(0, i + 1); + } + return toks; + } + if (kind === "const" && !isFunctionValue(toks)) return toks; + let depth = 0; + for (let i = 0; i < toks.length; i++) { + if (OPEN.has(toks[i])) depth += 1; + else if (CLOSE.has(toks[i])) depth -= 1; + else if (toks[i] === "=>" && depth === 0) return toks.slice(0, i + 1); + } + // A block body is the declaration's FINAL top-level `{…}` (a return-type literal before it + // stays in the head); no final block means no body here (an overload): all of it. + let end = toks.length - 1; + if (toks[end] === ";") end -= 1; + if (toks[end] !== "}") return toks; + depth = 0; + for (let i = end; i >= 0; i--) { + if (toks[i] === "}") depth += 1; + else if (toks[i] === "{" && --depth === 0) return toks.slice(0, i); + } + return toks; +} + /** - * A dependency's contract fingerprint: the hash of its DECLARATION as the source reads now - * (the declaration line through the opening brace/arrow, whitespace-normalized) — what a - * caller relies on. `null` when the atlas does not know the symbol or the file is unreadable. - * With several same-name definitions the first by (file, line) is used, deterministically. - * @param {string} root - * @param {any} atlas - * @param {string} name - * @returns {string|null} + * A dependency's contract, or why none can be given. + * @param {string} root @param {any} atlas @param {string} name + * @param {string|null} file the module the artifact's import binds `name` to (null: unbound) + * @returns {{sig: string}|{gone: string}|{unknown: string}} */ -export function depContract(root, atlas, name) { - const sym = (atlas?.symbols ?? []) - .filter((x) => x.name === name || x.qname === name) - .sort((a, b) => - a.file < b.file ? -1 : a.file > b.file ? 1 : (a.line ?? 0) - (b.line ?? 0), - )[0]; - if (!sym?.file || !sym.line) return null; +function contractOf(root, atlas, name, file) { + const cands = (atlas?.symbols ?? []).filter( + (x) => (x.name === name || x.qname === name) && !x.local, + ); + const files = [...new Set(cands.map((x) => x.file))]; + if (file && !files.includes(file)) return { gone: `no longer defined in ${file}` }; + if (!file && files.length !== 1) + return { + unknown: files.length + ? `defined in ${files.length} files and no import binding says which` + : "not in the atlas", + }; + const where = file ?? files[0]; let text; try { - text = readFileSync(join(root, sym.file), "utf8"); + text = readFileSync(join(root, where), "utf8"); } catch { - return null; + return { unknown: `${where} is unreadable` }; } - const decl = text - .split(/\r?\n/) - .slice(sym.line - 1, sym.line + 5) - .join(" "); - const cut = decl.search(/\{|=>/); - const head = (cut >= 0 ? decl.slice(0, cut) : decl).replace(/\s+/g, " ").trim(); - return head ? contentHash(`${sym.kind ?? ""}\u0000${head}`).slice(0, 16) : null; + const lines = text.split(/\r?\n/); + const lang = /\.pyi?$/.test(where) ? "py" : "js"; + const parts = []; + // Every same-file definition of the name (TS overloads), in source order. + for (const sym of cands.filter((x) => x.file === where).sort((a, b) => a.line - b.line)) { + if (!sym.line || !sym.endLine || sym.endLine < sym.line) + return { unknown: `the extent of ${name} in ${where} is not known` }; + const decl = lines.slice(sym.line - 1, sym.endLine).join("\n"); + const head = declarationHead(contractTokens(decl, lang), sym.kind, lang); + if (!head.length) return { unknown: `the declaration of ${name} is empty` }; + parts.push([sym.kind ?? "", head]); + } + return { sig: `${CONTRACT_PREFIX}${contentHash(JSON.stringify(parts)).slice(0, 16)}` }; +} + +/** + * A dependency's contract fingerprint (review F05/N08): a hash of its whole declaration minus + * its body (see above), prefixed with the contract version. `file` pins the definition the + * caller's import binds to; without it the name must be defined in exactly one file. `null` + * when the contract cannot be established — the caller must then treat it as unknown. + * @param {string} root + * @param {any} atlas + * @param {string} name + * @param {{file?: string|null}} [opts] + * @returns {string|null} + */ +export function depContract(root, atlas, name, { file = null } = {}) { + const r = contractOf(root, atlas, name, file); + return "sig" in r ? r.sig : null; } /** @@ -278,11 +467,22 @@ export function revalidate(artifact, atlas, { root = null } = {}) { if (deps.length) unknown.push("dependencies (no fresh atlas)"); } else { for (const d of deps) if (!atlasHas(atlas, d)) missing.push(d); + const sources = artifact?.body?.depSources ?? {}; for (const [name, sig] of Object.entries(artifact?.body?.depContracts ?? {})) { if (missing.includes(name)) continue; - const now = root ? depContract(root, atlas, name) : null; - if (now === null) unknown.push(`contract of ${name}`); - else if (now !== sig) changed.push(name); + if (typeof sig !== "string" || !sig.startsWith(CONTRACT_PREFIX)) { + unknown.push( + `contract of ${name} (${typeof sig === "string" ? "recorded in an older format" : "not established at mint"})`, + ); + continue; + } + if (!root) { + unknown.push(`contract of ${name} (no repo root)`); + continue; + } + const now = contractOf(root, atlas, name, sources[name] ?? null); + if ("unknown" in now) unknown.push(`contract of ${name} (${now.unknown})`); + else if ("gone" in now || now.sig !== sig) changed.push(name); } } if (missing.length) problems.push(`missing ${missing.join(", ")}`); @@ -328,7 +528,7 @@ export function lookup( spec, { slice = "", atlas = null, nowDay = 0, sim = null, root = null } = {}, ) { - const { key, sketch: qs, keySketch: qk } = fingerprint(spec, slice); + const { key, digest, sketch: qs, keySketch: qk } = fingerprint(spec, slice); const reasons = []; const invalidated = []; const artifacts = claims.filter( @@ -364,14 +564,13 @@ export function lookup( invalidated, }); - // 1. exact: the same task (LOSSLESS identity key, current key version), same graph-slice - // context. An empty key is not an identity, so a spec that normalizes to nothing never - // matches anything. + // 1. exact: the same task, byte for byte (the digest of the spec as given, current key + // version), same graph-slice context. A blank spec is not an identity: it never matches. for (const c of artifacts) { if ( - key && + key.trim() && c.body.keyV === KEY_VERSION && - c.body.key === key && + c.body.keyHash === digest && (c.body.slice ?? "") === slice && proved(c, "exact") ) { @@ -541,10 +740,27 @@ const EXPORT_RES = [ ]; const IMPORT_RE = /import\s+\{([^}]+)\}\s+from\s+["']\.{1,2}\//g; +/** The files a file's named imports BIND to, by imported name (review N08): the atlas resolves + * each import structurally (relative spec → file → that file's definition), so this is the + * module identity of the dependency, not a global name guess. @returns {Map>} */ +function importBindings(atlas, relPath) { + const nodes = new Map((atlas?.nodes ?? []).map((n) => [n.id, n])); + const out = new Map(); + for (const e of atlas?.edges ?? []) { + if (e.kind !== "imports" || !e.resolved || nodes.get(e.source)?.file !== relPath) continue; + const def = nodes.get(e.target); + if (!def?.file || def.kind === "module") continue; + if (!out.has(def.name)) out.set(def.name, new Set()); + out.get(def.name).add(def.file); + } + return out; +} + /** * The verifiable pointer + structural facts of a real file: `{path, sha256}`, exports, the - * codebase symbols it imports, and — given an atlas — each dependency's contract - * fingerprint, so a later signature change invalidates the artifact (review F05). + * codebase symbols it imports, and — given an atlas — each dependency's contract fingerprint + * and the module it is imported from, so a later contract change invalidates the artifact + * (review F05/N08). A dependency whose contract cannot be established is recorded as `null`. * @param {string} root * @param {string} relPath * @param {{atlas?: any}} [opts] @@ -563,18 +779,26 @@ export function describeFile(root, relPath, { atlas = null } = {}) { .filter(Boolean), ); const uniqueDeps = [...new Set(deps)]; - /** @type {Record} */ + /** @type {Record} */ const depContracts = {}; - if (atlas) + /** @type {Record} */ + const depSources = {}; + if (atlas) { + const bound = importBindings(atlas, relPath); for (const d of uniqueDeps) { - const sig = depContract(root, atlas, d); - if (sig) depContracts[d] = sig; + const files = [...(bound.get(d) ?? [])]; + // One bound module pins the definition; none (or an aliased pair) falls back to a name + // that must be unique repo-wide — else the contract is recorded as not established. + if (files.length === 1) depSources[d] = files[0]; + depContracts[d] = depContract(root, atlas, d, { file: depSources[d] ?? null }); } + } return { code: { path: relPath, sha256: contentHash(text) }, iface: [...new Set(iface)], deps: uniqueDeps, depContracts, + depSources, lang: relPath.split(".").pop() ?? "", }; } diff --git a/src/router/index.js b/src/router/index.js index 4cb78771..1798a7a3 100644 --- a/src/router/index.js +++ b/src/router/index.js @@ -141,9 +141,10 @@ export function routeUniversal(root, task, opts = {}) { applicable: unmapped.length === 0, unmapped, pSuccess: pick.p, - // EXPECTED cost under the fit (not a cap); the worst case runs every attempt. + // EXPECTED cost under the fit (not a cap), and the modeled cost if every attempt runs — + // also an estimate, never a bound on what a run can bill. expectedCost: pick.cost, - maxPossibleCost: pick.maxPossibleCost, + estimatedCostIfAllAttemptsRun: pick.estimatedCostIfAllAttemptsRun, minimumExpectedCost: pick.minimumExpectedCost, objective, target: pick.target, @@ -208,6 +209,41 @@ const outcomeSpec = (nFeatures) => }, }); +/** + * An outcome row's provenance, DERIVED — never read from the row (review N07): the outcomes + * file is user-editable, merged and replayed, so a stored `provenance: "verify-event"` label + * proves nothing. A row is "verify-event" only when its `verifyRunId` names an AUTHENTICATED + * verifier event in this checkout (readVerifyEvents: the MAC verifies under this machine's + * key), that run's verdict is pass/fail and agrees with the row's `passed`, and no earlier + * attempt already cites the same run (one verifier run backs one outcome — a replayed or + * copied row cannot multiply it). Anything else is "self-reported", with the reason in `note`. + * A verifier event authenticates the pass/fail VERDICT only: which model produced the patch, + * and what it cost, stay self-reported whatever the provenance. + * @param {any} row + * @param {Map} runs authenticated events by run id + * @param {Map} claimed run id → the attempt that cites it first + * @param {string} attempt this row's identity (attempt id, or a content key for legacy rows) + * @returns {{provenance: "self-reported"|"verify-event", note?: string}} + */ +function deriveProvenance(row, runs, claimed, attempt) { + const id = row.verifyRunId; + if (!id) return { provenance: "self-reported" }; + const self = (note) => ({ provenance: /** @type {const} */ ("self-reported"), note }); + const run = runs.get(id); + if (!run) return self(`verify run ${id} is not an authenticated verifier event in this checkout`); + const verdict = run.status === "PASS" ? true : run.status === "FAIL" ? false : null; + if (verdict === null) return self(`verify run ${id} is ${run.status}, not a pass/fail verdict`); + if (verdict !== row.passed) + return self( + `verify run ${id} says ${run.status}, the row says ${row.passed ? "pass" : "fail"}`, + ); + const first = claimed.get(id); + if (first !== undefined && first !== attempt) + return self(`verify run ${id} already backs attempt ${first}`); + claimed.set(id, attempt); + return { provenance: "verify-event" }; +} + /** * Record the outcome of one attempt (the only evidence the router learns from). The task text * is not stored: only its hash and features. Validated before it is written (review A04): a @@ -215,10 +251,14 @@ const outcomeSpec = (nFeatures) => * a feature vector of the wrong length is refused. * * `passed` is SELF-REPORTED by the caller unless `verifyRunId` names a `forge verify` run in - * this checkout's verifier-event log whose verdict agrees (PASS ⇔ passed, FAIL ⇔ failed) — - * then the row is `provenance: "verify-event"` (A01). `attemptId` makes recording idempotent: - * the same attempt recorded twice (a retry, a replayed file) counts once; omit it and a fresh - * id is minted. + * this checkout's verifier-event log whose verdict agrees (PASS ⇔ passed, FAIL ⇔ failed) and + * whose event is AUTHENTICATED — then the row is `provenance: "verify-event"` (A01). An event + * that exists but is not authenticated (no evidence key could be read or created, or its MAC + * does not verify) never earns that label (review N06): the row is recorded as self-reported, + * with the reason in `provenanceNote`. A run that already backs another attempt is refused. + * `attemptId` makes recording idempotent: the same attempt recorded twice (a retry, a + * replayed file) counts once; omit it and a fresh id is minted. Readers re-derive provenance + * from the events either way (readOutcomes), so the stored label is informational. * @param {string} root * @param {{task: string, model: string, passed: boolean, cost?: number|null, * features?: number[]|null, attemptId?: string|null, verifyRunId?: string|null, @@ -241,7 +281,12 @@ export function recordOutcome( throw new Error("recordOutcome needs model and passed (boolean)"); if (!registry.models.some((m) => m.id === model)) throw new Error(`unknown model "${model}" — not in the registry (\`forge route models\`)`); + const id = attemptId ?? randomUUID(); + const existing = readOutcomes(root); + const prev = existing.find((o) => o.attemptId === id); + if (prev) return { ...prev, duplicate: true }; let provenance = "self-reported"; + let provenanceNote = null; if (verifyRunId) { const run = readVerifyEvents(root).find((e) => e.runId === verifyRunId); if (!run) throw new Error(`no verify run ${verifyRunId} in .forge/verify-events.jsonl`); @@ -252,12 +297,16 @@ export function recordOutcome( throw new Error( `verify run ${verifyRunId} says ${run.status}, not ${passed ? "pass" : "fail"}`, ); - provenance = "verify-event"; - } - const id = attemptId ?? randomUUID(); - if (readOutcomes(root).some((o) => o.attemptId === id)) { - const prev = readOutcomes(root).find((o) => o.attemptId === id); - return { ...prev, duplicate: true }; + const other = existing.find( + (o) => o.verifyRunId === verifyRunId && o.provenance === "verify-event", + ); + if (other) + throw new Error( + `verify run ${verifyRunId} already backs attempt ${other.attemptId} — one verifier run backs one outcome`, + ); + if (run.authenticated) provenance = "verify-event"; + else + provenanceNote = `verify run ${verifyRunId} is not authenticated (no evidence key could be read or created, or its MAC does not verify) — recorded as self-reported`; } const row = { attemptId: id, @@ -269,6 +318,7 @@ export function recordOutcome( cost: cost === null || cost === undefined ? null : Number(cost), provenance, ...(verifyRunId ? { verifyRunId } : {}), + ...(provenanceNote ? { provenanceNote } : {}), }; const v = validate(row, outcomeSpec(featureCount(root)), "outcome"); if (!v.ok) throw new Error(`invalid outcome: ${v.errors.join("; ")}`); @@ -279,18 +329,32 @@ export function recordOutcome( } /** - * The recorded outcomes, validated and deduplicated: a row that fails the outcome schema is - * skipped (and counted in `readOutcomes.lastInvalid`); rows sharing an `attemptId` count once; - * legacy rows without one are deduplicated by their exact content, so a replayed or - * union-merged file never multiplies training evidence. + * The recorded outcomes, validated, deduplicated and with provenance RE-DERIVED from this + * checkout's authenticated verifier events (deriveProvenance, review N07): a row that fails + * the outcome schema is skipped (counted in `readOutcomes.lastInvalid`); rows sharing an + * `attemptId` count once; legacy rows without one are deduplicated by their exact content, so + * a replayed or union-merged file never multiplies training evidence. A row whose stored label + * said "verify-event" but whose run is missing, unauthenticated, disagrees, or already backs + * another attempt is read as "self-reported" with a `provenanceNote`, and counted in + * `readOutcomes.lastDowngraded`. * @param {string} root */ export function readOutcomes(root) { + readOutcomes.lastInvalid = 0; + readOutcomes.lastDowngraded = 0; if (!existsSync(OUTCOMES(root))) return []; const spec = outcomeSpec(featureCount(root)); + const runs = new Map( + readVerifyEvents(root) + .filter((e) => e.authenticated) + .map((e) => [e.runId, e]), + ); + /** @type {Map} */ + const claimed = new Map(); const seen = new Set(); const out = []; let invalid = 0; + let downgraded = 0; for (const line of readFileSync(OUTCOMES(root), "utf8").split("\n")) { if (!line.trim()) continue; let row; @@ -309,12 +373,17 @@ export function readOutcomes(root) { : `row:${createHash("sha256").update(JSON.stringify(row)).digest("hex")}`; if (seen.has(key)) continue; seen.add(key); - out.push(row); + const { provenanceNote: _stored, ...rest } = row; + const d = deriveProvenance(row, runs, claimed, row.attemptId ?? key); + if (row.provenance === "verify-event" && d.provenance !== "verify-event") downgraded++; + out.push({ ...rest, provenance: d.provenance, ...(d.note ? { provenanceNote: d.note } : {}) }); } readOutcomes.lastInvalid = invalid; + readOutcomes.lastDowngraded = downgraded; return out; } readOutcomes.lastInvalid = 0; +readOutcomes.lastDowngraded = 0; /** * Refit on the project's recorded outcomes with the shipped fit as the prior mean (Bayesian @@ -381,8 +450,11 @@ export function fitRouter( prior: shipped.provenance, local: { outcomes: outcomes.length - skipped, - // How much of the local evidence is tied to a verifier event vs self-reported (A01). + // How much of the local evidence is tied to an AUTHENTICATED verifier event vs + // self-reported (A01/N07 — derived by readOutcomes, never read from the row). The event + // vouches for the pass/fail verdict only; model and cost stay self-reported. verifiedOutcomes: verified, + verifiedFields: ["passed"], selfReportedOutcomes: outcomes.length - skipped - verified, skipped, tasks: data.tasks.length, diff --git a/src/router/policy.js b/src/router/policy.js index f7d9d6c4..4bf50c57 100644 --- a/src/router/policy.js +++ b/src/router/policy.js @@ -21,9 +21,10 @@ // the least-bad cascade is still computed, but the result says `feasible: false` with a // `reason`, `budgetMet: false` for a budget objective, and `minimumExpectedCost` — the caller // must choose to fall back; nothing reads as "within budget". A budget constrains EXPECTED -// cost only: `maxPossibleCost` (every attempt in the cascade runs) is the worst case, and the -// ACTUAL charge is whatever the attempts cost — enforce a hard cap at execution time if one is -// promised. The cost formula also assumes an attempt's cost does not depend on earlier +// cost only. `estimatedCostIfAllAttemptsRun` is the sum of every attempt's EXPECTED cost — +// a modeled figure for the path where each stage runs, not a bound on what a stochastic run can +// bill (it was called `maxPossibleCost`, which overstated it). The ACTUAL charge is whatever the +// attempts cost — enforce a hard cap at execution time if one is promised. The cost formula also assumes an attempt's cost does not depend on earlier // attempts having failed (E[cost_i | earlier failed, x] = E[cost_i | x]); hard residual tasks // may cost more, so treat expected cost as an estimate under that assumption. @@ -113,7 +114,7 @@ export function choose(nodes, costs, candidates, objective, maxDepth) { const targetMet = target === null ? null : best.p >= target - eps; const budgetMet = objective.kind === "budget" ? best.cost <= objective.budget + eps : null; const minimumExpectedCost = Math.min(...single.map((x) => x.cost)); - const maxPossibleCost = best.seq.reduce((sum, m) => sum + costs[m], 0); + const estimatedCostIfAllAttemptsRun = best.seq.reduce((sum, m) => sum + costs[m], 0); const feasible = targetMet !== false && budgetMet !== false; const reason = feasible ? undefined @@ -128,7 +129,7 @@ export function choose(nodes, costs, candidates, objective, maxDepth) { feasible, ...(reason ? { reason } : {}), minimumExpectedCost, - maxPossibleCost, + estimatedCostIfAllAttemptsRun, bestSingle: { model: bestSingle.m, p: bestSingle.p, cost: bestSingle.cost }, evaluated, }; diff --git a/src/semantic_guard.js b/src/semantic_guard.js index 404a3a1a..a189595c 100644 --- a/src/semantic_guard.js +++ b/src/semantic_guard.js @@ -8,8 +8,10 @@ // // Deliberately conservative and deterministic: the features are the tokens that carry // behaviour — operators, numbers (with units), quoted literals, code identifiers and paths -// (case kept), and polarity/negation words. A false conflict costs a human glance; a missed -// one serves or keeps the opposite rule. Pure — no I/O. +// (case AND code points kept: no Unicode fold, since `"é"` composed and decomposed are two +// different strings to a program), polarity/negation words, and the whitespace that can be +// data (`layout`, review N01). A false conflict costs a human glance; a missed one serves or +// keeps the opposite rule. Pure — no I/O. // Polarity is judged two ways, so that emphasis ("run X" vs "always run X") is not a flip // but a real reversal is: @@ -95,14 +97,88 @@ export function trimEdges(raw) { return cps.slice(a, b).join(""); } +// Whitespace that can be DATA (review N01). Similarity is layout-blind by design, and a +// reworded instruction moves whitespace around harmlessly — but a fenced code block's bytes, +// an indented line of a text that carries code (Python, YAML, a Makefile tab), and any run of +// whitespace other than one space beside a code token (`/a b/`, `x\t= 1`) can change what the +// code does. Those are compared verbatim; whitespace between two plain words never is, and a +// line break in prose reads as one space. (Whitespace INSIDE a quoted literal is already part +// of that literal.) +const FENCE_RE = /^[ \t]*(`{3,}|~{3,})/; +const CODE_LINE_RE = /[(){}[\];=<>|&]|:[ \t]*$/; +const PLAIN_WORD_RE = /^[\p{L}\p{N}_'’-]*$/u; +const LIT_MARK = "\u27e8lit\u27e9"; // a masked literal: a code token with no inner whitespace +const OPENERS = new Set([..."(\"'‘“["]); +const CLOSERS = new Set([...".,;:!?)\"'’”]"]); +/** Strip opening brackets/quotes and closing punctuation from a token's edges — a code-point + * scan, linear on any input — then what is left of a plain word is letters and digits; + * anything else marks a code token. */ +const isPlainWord = (tok) => { + const cps = [...tok]; + let a = 0; + let b = cps.length; + while (a < b && OPENERS.has(cps[a])) a++; + while (b > a && CLOSERS.has(cps[b - 1])) b--; + return PLAIN_WORD_RE.test(cps.slice(a, b).join("")); +}; + +/** + * The layout features of a text (review N01): every fenced block and every indented line of a + * code-bearing text, verbatim, plus every whitespace run other than a single space that sits + * next to a code token, with its neighbours. Sorted; JSON-quoted so whitespace is visible. + * @param {string} text + * @returns {string[]} + */ +export function layoutFeatures(text) { + const out = []; + const rest = []; + let fence = ""; + let block = []; + for (const line of String(text ?? "").split("\n")) { + const m = FENCE_RE.exec(line); + if (fence) { + block.push(line); + if ( + m && + m[1][0] === fence[0] && + m[1].length >= fence.length && + !line.slice(m[0].length).trim() + ) { + out.push(`fence ${JSON.stringify(block.join("\n"))}`); + fence = ""; + } + } else if (m) { + fence = m[1]; + block = [line]; + } else rest.push(line); + } + if (fence) out.push(`fence ${JSON.stringify(block.join("\n"))}`); // unclosed: code to the end + const lines = rest.join("\n").replace(LITERAL_RE, LIT_MARK).split("\n"); + if (lines.some((l) => CODE_LINE_RE.test(l))) + for (const l of lines) + if (/^[ \t]+\S/.test(l)) out.push(`indent ${JSON.stringify(l.trimEnd())}`); + const parts = lines + .map((l) => l.trim()) + .filter(Boolean) + .join(" ") + .split(/(\s+)/); + for (let i = 1; i + 1 < parts.length; i += 2) { + if (parts[i] === " ") continue; + const [left, right] = [parts[i - 1], parts[i + 1]]; + if (!isPlainWord(left) || !isPlainWord(right)) + out.push(`space ${JSON.stringify(`${left}${parts[i]}${right}`)}`); + } + return sorted(out); +} + /** * The behaviour-carrying features of a text. * @param {string} text * @returns {{operators: string[], numbers: string[], literals: string[], - * identifiers: string[], paths: string[], polarity: string[]}} + * identifiers: string[], paths: string[], polarity: string[], layout: string[]}} */ export function criticalFeatures(text) { - const s = String(text ?? "").normalize("NFC"); + const s = String(text ?? ""); /** @type {string[]} */ const literals = s.match(LITERAL_RE) ?? []; // Literals are compared whole; strip them before scanning the rest so a quoted ">=" or a @@ -132,6 +208,7 @@ export function criticalFeatures(text) { identifiers: sorted(identifiers), paths: sorted(paths), polarity: sorted(polarity), + layout: layoutFeatures(s), }; } @@ -165,6 +242,7 @@ export const ALL_KINDS = /** @type {const} */ ([ "literals", "identifiers", "paths", + "layout", ]); export const FLIP_KINDS = /** @type {const} */ (["polarity", "operators", "numbers", "literals"]); @@ -193,14 +271,35 @@ export function semanticConflicts(a, b, { kinds = ALL_KINDS } = {}) { return out; } -/** True when neither text changes polarity, operators, numbers, literals, identifiers or - * paths relative to the other. Similar-and-conflicting texts are NOT the same instruction. +/** True when neither text changes polarity, operators, numbers, literals, identifiers, + * paths or code layout relative to the other. Similar-and-conflicting texts are NOT the same instruction. * @param {string} a @param {string} b @param {{kinds?: readonly string[]}} [opts] */ export const sameSemantics = (a, b, opts) => semanticConflicts(a, b, opts).length === 0; +/** The exact-duplicate key of a statement (review N02): whitespace runs collapsed and trailing + * sentence punctuation dropped — nothing else (case, quotes, inner punctuation all stay). */ +export const statementKey = (s) => { + const t = String(s ?? "") + .trim() + .replace(/\s+/g, " "); + let end = t.length; + while (end > 0 && ".!;".includes(t[end - 1])) end--; // a scan: `[.!;]+$` backtracks on runs + return t.slice(0, end); +}; + +/** + * The SAME statement, not a similar one (review N02): equal statement keys, with the guard + * confirming no behaviour-bearing difference hid in the whitespace (inside a literal, in code + * layout). The only equality that may merge or archive a rule automatically. Similarity — + * lexical or embedded — can only PROPOSE: "allow admins, deny guests" and "deny admins, allow + * guests" keep every token, polarity words included, so no token-level check separates them. + * @param {string} a @param {string} b + */ +export const sameStatement = (a, b) => statementKey(a) === statementKey(b) && sameSemantics(a, b); + /** One-line human description of a conflict list: `polarity: enable ≠ disable; …`. */ export function describeConflicts(conflicts) { - return conflicts - .map((c) => `${c.kind}: ${c.a.join(" ") || "∅"} ≠ ${c.b.join(" ") || "∅"}`) - .join("; "); + // A layout item can be a whole code block: shown truncated, compared in full. + const show = (xs) => xs.map((x) => (x.length > 72 ? `${x.slice(0, 71)}…` : x)).join(" ") || "∅"; + return conflicts.map((c) => `${c.kind}: ${show(c.a)} ≠ ${show(c.b)}`).join("; "); } diff --git a/src/stack.js b/src/stack.js index 24531f44..e88c46e8 100644 --- a/src/stack.js +++ b/src/stack.js @@ -327,8 +327,9 @@ const MONO_MAX_DEPTH = 3; // deepest nested dir considered (e.g. apps/web, packa const MONO_SCAN_BUDGET = 200; // hard ceiling on dirs stat-ed — bounds cost on large trees const MONO_MAX_ROOTS = 50; // most nested package roots surfaced -// Zero-dep: pull the `packages:` list from a pnpm-workspace.yaml. Ignores negations (`!…`). -function parsePnpmPackages(text) { +// Zero-dep: pull the `packages:` list from a pnpm-workspace.yaml. Negations (`!…`) are +// dropped unless `negations` is set (membership needs them: `!packages/legacy` is NOT run). +function parsePnpmPackages(text, { negations = false } = {}) { const out = []; let inBlock = false; for (const raw of text.split(/\r?\n/)) { @@ -341,7 +342,7 @@ function parsePnpmPackages(text) { const m = /^\s*-\s*(.+?)\s*$/.exec(line); if (m) { const v = m[1].trim().replace(/^["']|["']$/g, ""); - if (v && !v.startsWith("!")) out.push(v); + if (v && (negations || !v.startsWith("!"))) out.push(v); } else if (/^\S/.test(line)) { inBlock = false; // a new top-level key ends the list } @@ -427,16 +428,442 @@ export function matchesWorkspaceGlob(glob, rel) { return new RegExp(`^${re}$`).test(stripTrailingSlashes(rel)); } -// A root test script that runs EVERY workspace's suite itself — so a nested package that is a -// declared workspace member is covered by the root run and must not be run twice. -const RECURSIVE_TEST_RE = - /(?:^|\s)(?:--workspaces|-ws|--recursive|-r)(?:\s|$)|\bworkspaces\s+foreach\b|\bturbo\s+(?:run\s+)?test\b|\blerna\s+run\s+test\b|\bnx\s+run-many\b/; +/** + * Is `rel` a member of a workspace glob list? Included by some positive glob and excluded by + * no `!negation` — the way npm, yarn and pnpm read their lists. Pure. + * @param {string[]} globs @param {string} rel + */ +export function isWorkspaceMember(globs, rel) { + let member = false; + for (const g of globs) { + if (typeof g !== "string") continue; + if (g.startsWith("!")) { + if (matchesWorkspaceGlob(g.slice(1), rel)) return false; + } else if (!member && matchesWorkspaceGlob(g, rel)) member = true; + } + return member; +} + +// --------------------------------------------------------------------------- +// Recursive workspace test runs (review N03). A root `test` script covers the workspaces only +// when it PROVABLY runs every member's own `test` script and that run's failure reaches the +// script's exit status. The old check matched a flag token anywhere in the text, so Node's +// preload flag (`node -r ./setup.cjs --test`) read as `pnpm -r`: a failing workspace was never +// executed and verify said PASS. The script is now read as shell STRUCTURE — commands split at +// `&&` `||` `;` `|` `&`, quotes resolved, each command's executable identified through +// env/cross-env/npx/exec wrappers — and only a known, UNFILTERED recursive run of `test` +// counts: +// npm test|run test --workspaces|-ws no -w/--workspace/--prefix +// pnpm -r|--recursive test|run test no --filter/-F/--resume-from/-C/-w/--no-bail +// yarn workspaces run test (v1) +// yarn workspaces foreach [-A|-W] [run] test no --include/--exclude/--since/--from/-R/--no-private +// turbo run test [other tasks] no --filter/-F/--affected/--since/--scope/--dry-run/--continue +// lerna run test no --scope/--ignore/--since/--no-private/--no-bail +// nx run-many -t test no --projects/-p/--exclude +// — directly or through at most four `npm run ` hops. Its status must reach the +// script's: the last command of its pipeline, in the script's final list, not backgrounded, +// with no `||` right before it or anywhere after it. Anything else — including a construct +// this reader does not model (`$VAR`, `$(…)`, subshells, here-docs) — is NOT established, and +// the planner runs each workspace suite in its own directory; only `verify.workspaces: +// "root"` can declare otherwise. Which packages the run reaches is read from the tool's OWN +// workspace list (npm/yarn: package.json, pnpm: pnpm-workspace.yaml, lerna: lerna.json), +// negations included. +// --------------------------------------------------------------------------- + +/** + * Read a package-script command line the way `sh` splits it — for RECOGNITION only, nothing + * is executed. Each command is its words (quotes and backslashes resolved, redirections + * dropped) plus the control operator that follows it (`&&`, `||`, `;`, `|`, `&`, or null at + * the end). `null` when the line uses a construct this reader does not model: parameter + * expansion, command substitution, subshells or `{ }` groups, here-docs, a dangling quote. + * @param {string} line + * @returns {{words: string[], op: string|null}[]|null} + */ +export function shellCommands(line) { + const s = String(line); + /** @type {{words: string[], op: string|null}[]} */ + const cmds = []; + /** @type {string[]} */ + let words = []; + /** @type {string|null} */ + let word = null; // null = no word in progress, so an empty quoted "" is still a word + let quoted = false; + let redirect = false; // the next word is a redirection target, not an argument + const endWord = () => { + if (word === null) return true; + const w = word; + word = null; + if (!quoted && (w === "{" || w === "}")) return false; // a `{ …; }` group + quoted = false; + if (redirect) redirect = false; + else words.push(w); + return true; + }; + const endCommand = (op) => { + if (!endWord() || redirect) return false; // `>` with no target + if (!words.length) return op === ";" || op === "\n" || op === null; // blank line / trailing `;` + cmds.push({ words, op: op === "\n" ? ";" : op }); + words = []; + return true; + }; + for (let i = 0; i < s.length; i++) { + const c = s[i]; + if (c === "'") { + const j = s.indexOf("'", i + 1); + if (j < 0) return null; + word = (word ?? "") + s.slice(i + 1, j); + quoted = true; + i = j; + } else if (c === '"') { + let out = ""; + let j = i + 1; + for (; j < s.length && s[j] !== '"'; j++) { + if (s[j] === "$" || s[j] === "`") return null; + if (s[j] === "\\" && j + 1 < s.length && '"\\$`\n'.includes(s[j + 1])) out += s[++j]; + else out += s[j]; + } + if (j >= s.length) return null; + word = (word ?? "") + out; + quoted = true; + i = j; + } else if (c === "\\") { + if (i + 1 >= s.length) return null; + if (s[i + 1] !== "\n") { + word = (word ?? "") + s[i + 1]; + quoted = true; + } + i += 1; + } else if (c === "$" || c === "`" || c === "(" || c === ")") return null; + else if (c === "#" && word === null) { + while (i + 1 < s.length && s[i + 1] !== "\n") i += 1; // a comment runs to end of line + } else if (c === "\n") { + if (!endCommand("\n")) return null; + } else if (c === " " || c === "\t" || c === "\r") { + if (!endWord()) return null; + } else if (c === "<" || c === ">") { + // Redirection. A pending all-digit word is its fd (`2>`), not an argument. + if (word !== null && /^\d+$/.test(word) && !quoted) word = null; + else if (!endWord()) return null; + if (redirect) return null; + if (c === "<" && s[i + 1] === "<") return null; // here-doc + if (s[i + 1] === ">" || s[i + 1] === "&" || s[i + 1] === "|") i += 1; + redirect = true; + } else if (c === "&" || c === "|" || c === ";") { + const two = s.slice(i, i + 2); + if (c === "&" && s[i + 1] === ">") { + if (!endWord()) return null; // `&>file` — a redirection, not an operator + i += s[i + 2] === ">" ? 2 : 1; + redirect = true; + continue; + } + if (two === ";;") return null; + const op = two === "&&" || two === "||" ? two : two === "|&" ? "|" : c; + if (op.length === 2 || two === "|&") i += 1; + if (!endCommand(op)) return null; + } else { + word = (word ?? "") + c; + } + } + if (!endCommand(null)) return null; + const last = cmds[cmds.length - 1]; + if (last && (last.op === "&&" || last.op === "||" || last.op === "|")) return null; // dangling + if (last?.op === ";") last.op = null; + return cmds; +} + +/** + * The commands whose exit status reaches the whole line's (no `set -e` modelled): those in + * the FINAL list (after the last `;`/`&`), last in their pipeline (no `pipefail`), with no + * `||` right before them (they might be skipped on success) or anywhere after them (their + * failure would be absorbed). A backgrounded final command reaches nothing. + * @param {{words: string[], op: string|null}[]} cmds + */ +function statusCarriers(cmds) { + if (!cmds.length || cmds[cmds.length - 1].op === "&") return []; + let start = 0; + for (let i = 0; i < cmds.length - 1; i++) + if (cmds[i].op === ";" || cmds[i].op === "&") start = i + 1; + const list = cmds.slice(start); + const out = []; + for (let i = 0; i < list.length; i++) { + if (list[i].op === "|") continue; // not the last command of its pipeline + let first = i; + while (first > 0 && list[first - 1].op === "|") first -= 1; + if (first > 0 && list[first - 1].op === "||") continue; + if (list.slice(i).some((x) => x.op === "||")) continue; + out.push(list[i]); + } + return out; +} + +const ENV_ASSIGN = /^[A-Za-z_][A-Za-z0-9_]*=/; +// Wrappers whose job is to find and run another binary. +const TOOL_BINS = new Set(["turbo", "lerna", "nx"]); +const NPX_OK = new Set([ + "--yes", + "-y", + "--no", + "--no-install", + "--quiet", + "-q", + "--prefer-offline", +]); +/** `./node_modules/.bin/turbo.cmd` → `turbo`. */ +const binName = (w) => + String(w) + .split(/[\\/]/) + .pop() + .replace(/\.(?:cmd|exe|ps1|js|cjs|mjs)$/i, ""); + +/** Strip env assignments and runner wrappers (env, cross-env, npx, bunx, npm exec, pnpm/yarn + * exec|dlx, `yarn turbo`) down to the executable that does the work. `null` when a wrapper + * is used in a way that could change what runs (`npx -c`, `npm exec --workspaces`…). */ +function unwrapCommand(words) { + const w = [...words]; + for (let hop = 0; hop < 6; hop++) { + while (w.length && ENV_ASSIGN.test(w[0])) w.shift(); + if (!w.length || w[0] === "!") return null; // `! cmd` inverts the status + const bin = binName(w[0]); + const next = w[1]; + if (bin === "env" || bin === "cross-env") { + w.shift(); + while (w.length && w[0].startsWith("-")) { + if (w[0] === "-u" || w[0] === "--unset") w.shift(); + w.shift(); + } + continue; + } + if (bin === "npx" || bin === "bunx" || (bin === "bun" && next === "x")) { + w.shift(); + if (bin === "bun") w.shift(); + while (w.length && w[0].startsWith("-")) { + const o = w.shift(); + if (o === "-p" || o === "--package") w.shift(); + else if (o === "--") break; + else if (!NPX_OK.has(o) && !o.startsWith("--package=")) return null; + } + continue; + } + if (bin === "npm" && (next === "exec" || next === "x")) { + w.splice(0, 2); + while (w.length && w[0].startsWith("-")) { + const o = w.shift(); + if (o === "--") break; + if (!NPX_OK.has(o) && !o.startsWith("--package=")) return null; + } + continue; + } + if ((bin === "pnpm" || bin === "yarn") && (next === "exec" || next === "dlx")) { + w.splice(0, 2); + if (w[0] === "--") w.shift(); + if (w[0]?.startsWith("-")) return null; + continue; + } + if ((bin === "pnpm" || bin === "yarn") && next && TOOL_BINS.has(binName(next))) { + w.shift(); + continue; + } + return { bin, args: w.slice(1), words: w }; + } + return null; +} + +// Options that make a run cover a SUBSET of the workspaces, another root, or not propagate +// every failure — any of them and the run is not a whole-workspace verdict. +const NPM_NARROW = /^(?:-w|--workspace|-C|--prefix)(?:=|$)/; +const PNPM_NARROW = + /^(?:--filter|-F|--filter-prod|--resume-from|-C|--dir|-w|--workspace-root|--test-pattern|--changed-files-ignore-pattern|--no-bail)(?:=|$)/; +const YARN_NARROW = /^(?:--include|--exclude|--since|--from|--recursive|--no-private)(?:=|$)/; +const TURBO_NARROW = + /^(?:--filter|-F|--affected|--since|--scope|--ignore|--dry-run|--dry|--graph|--continue|--cwd)(?:=|$)/; +const LERNA_NARROW = + /^(?:--scope|--ignore|--since|--no-private|--no-bail|--exclude-dependents|--include-merged-tags)(?:=|$)/; +const NX_NARROW = + /^(?:--projects|-p|--exclude|--affected|--base|--head|--files|--uncommitted|--untracked)(?:=|$)/; +const TEST_ALIASES = new Set(["test", "t", "tst"]); +const RUN_ALIASES = new Set(["run", "run-script", "rum", "urn"]); +// pnpm builtins that are not script names (`pnpm