diff --git a/CHANGELOG.md b/CHANGELOG.md
index b5a2fb0..345cf8f 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -6,6 +6,57 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## [Unreleased]
+Fixes for the two findings of the 2026-09-27 recheck of v1.7.3 (Q01, Q02). The recheck
+confirmed that all eight N01–N08 reproduction cases are closed; its scripts
+(`reproduce_remaining.mjs`, `original-reproduce.mjs` and `edge-probes.mjs`, each with
+`--assert-fixed`) all pass.
+
+### Fixed
+
+- **A local program named like a package manager no longer covers the workspaces (Q02).** A
+ root `test` script of `./tools/npm test --workspaces` was credited as npm's recursive run
+ because the recognizer read only the file's basename, so a stub that exited 0 hid a failing
+ workspace behind a PASS. A tool is now recognized only by its bare name (the program the
+ script's PATH finds) or as its own install under `node_modules/.bin`. Any other path, such as
+ `./tools/npm`, `./scripts/pnpm.js` or `/usr/bin/env`, is an unknown program whatever its
+ name, so the members run on their own. `forge verify` reports the root run as "not credited"
+ and names the program.
+ - A command line may set only variables known to change neither the program nor the run
+ (`CI`, `NODE_ENV`, `FORCE_COLOR`, `NODE_OPTIONS` with memory or warning flags…). `PATH`,
+ `LD_PRELOAD`, `HOME`, a `NODE_OPTIONS` preload and any other variable keep the run from
+ being credited.
+ - Wrapper options that pick the binary, the directory or the environment are refused:
+ `npx -p`/`--package`, `env -i`, `env -C`/`--chdir`, and options on `cross-env`.
+ - A bare name must not be shadowed. The first `node_modules/.bin` entry on the script's PATH
+ (the package's own, then each parent directory's) must be the tool's own package's binary,
+ and a symlink must resolve into that package. A package manager or system program there is
+ a shim. On Windows, a same-named executable in the package root (`npm.cmd`) runs first, so
+ it counts as a shadow too.
+ - yarn must be a release: a `yarnPath` (or yarn 1's `yarn-path`) outside `.yarn/releases/`,
+ or a `packageManager` that fetches the manager from a URL, is refused.
+ - `.npmrc` and environment `node-options` must be inert, a `script-shell` that is a program
+ inside the project is not a shell, and nx plugins, which can redefine each project's `test`
+ target, make nx and lerna runs run the members on their own.
+ - A `verify.workspaces: "root"` declaration still covers every member and is still labelled
+ `declared`.
+- **A near reuse hit is no longer presented as the same task (Q01).** "Deny admins and allow
+ guests…" near-hit an artifact verified for "Allow admins and deny guests…" (MinHash 0.83),
+ and the CLI called it a "reworded match".
+ - Every hit now says what it establishes. `semanticEquivalence` is `"identical"` for an
+ exact hit (the byte-identical spec) and `"unverified"` for near and adapt, and
+ `requiresReview` is `true` for every non-exact hit. `forge reuse query --json` and the gate's
+ reuse summary carry both fields, and the CLI and gate describe a near hit as "a similar
+ task, equivalence unverified".
+ - The semantic guard adds a `binding` conflict kind. Each polarity, negation or direction
+ word binds to the next content word of its clause. Two texts conflict when a word both bind
+ is bound to opposite relations (allow→admins vs deny→admins, from→staging vs to→staging), or
+ when they use exactly the same words in a different order. Permission-subject swaps,
+ source/destination swaps and role swaps are therefore held at `adapt`. Consolidation and
+ compaction list such pairs as conflicts rather than proposals.
+- **The claims check no longer fails in a checkout that lacks newer release tags.** A claim
+ naming a release newer than every local tag, and not newer than `package.json`'s version, is
+ reported as unchecked, with a hint to run `git fetch --tags`, instead of failing the check.
+
## [1.7.3] - 2026-09-27
Fixes for the eight findings of the 2026-09-27 follow-up review (N01–N08) and its suggestions.
diff --git a/docs/GUIDE.md b/docs/GUIDE.md
index 0ef7537..0e38e84 100644
--- a/docs/GUIDE.md
+++ b/docs/GUIDE.md
@@ -705,11 +705,24 @@ conservative too:
- Options are allowlisted per tool. An unknown one (`--help`, `--dry-run`, an abbreviation
npm would expand) is not credited, and neither are words after `--`, which are forwarded
into every member's script.
-- A script that changes shell state (`cd`, `exit`, `trap`, `export`…) or sets a
- package-tool variable (`npm_config_*`) is not credited.
+- A tool is recognized by its bare name, the program the script's PATH finds, or as its own
+ install under `node_modules/.bin`. Any other path (`./tools/npm`, `./scripts/pnpm.js`,
+ `/usr/bin/env`) is an unknown program whatever its name, and is never credited.
+- A script that changes shell state (`cd`, `exit`, `trap`, `export`…) is not credited, and
+ the command may set only variables known to change neither the program nor the run (`CI`,
+ `NODE_ENV`, `FORCE_COLOR`, `NODE_OPTIONS` with memory or warning flags, the cache-bypass
+ variables…). `PATH`, `LD_PRELOAD`, `HOME`, `npm_config_*` and anything else are refused, and
+ so are wrapper options that pick the binary or the directory (`npx -p`, `env -C`).
+- A bare name must not be shadowed. The first `node_modules/.bin` entry on the script's PATH
+ (the package's own, then each parent directory's) must be the binary of the tool's own
+ installed package, and a symlink must resolve into it. A copy of a package manager or a
+ system program there is a shim. A same-named executable in the package root (`npm.cmd`)
+ counts too, because Windows runs it first. yarn must be a release: a `yarnPath` outside
+ `.yarn/releases/`, or a `packageManager` fetched from a URL, is refused.
- Configuration that narrows the run is honoured: `.npmrc` or environment
- `workspace`/`filter`/`script-shell`, lerna `command.run` filters, `.nxignore`, a
- redefined nx `test` target, and turbo per-package tasks.
+ `workspace`/`filter`/`script-shell`, a `node-options` that loads code, lerna `command.run`
+ filters, `.nxignore`, a redefined nx `test` target or nx plugins (which can define it), and
+ turbo per-package tasks.
- A cached result is not a run: turbo needs `--force` (or `cache: false` for `test`), and nx
and lerna need `--skip-nx-cache`.
- Only members of that tool's own workspace list count, read for the package manager in use.
@@ -718,11 +731,11 @@ conservative too:
The `coverage basis` line says what each package's verdict rests on: `measured` (its own
suite ran), `inferred` (the recognized root command) or `declared` (`workspaces: "root"`). A
-recognized root command that was not credited is printed as `root run not credited`, with the
-reason. forge trusts the repository's own tooling: a shim that replaces the package manager
-or the test runner is outside what it checks (one in `node_modules/.bin` shadowing
-npm/pnpm/yarn is refused). A `script-shell` that is not a shell (`/bin/true`) makes npm and
-pnpm suites `INCOMPLETE`. Fixture and test-data packages are never required. A test runner
+root command that was not credited, including one that names a program by its path, is
+printed as `root run not credited`, with the reason. Beyond these checks forge trusts the
+installed tools themselves: it does not audit what a genuine package's binary or the test
+runner does. A `script-shell` that is not a shell (`/bin/true`), or that is a program inside
+the project, makes npm and pnpm suites `INCOMPLETE`. Fixture and test-data packages are never required. A test runner
that is only a devDependency is not an obligation either; `forge stack` lists it as
`available`.
Tune this per repo under `verify` in `.forge/forge.config.json`:
@@ -1170,10 +1183,20 @@ unchanged.
never share a key. Artifacts minted before key version 3 never hit exact. Inline code the
ledger's storage would rewrite (CRLF line endings, non-NFC text) is refused at mint; mint
it from a file instead.
-- **Near must also agree on behaviour.** A reworded match is offered as `near` only when the
+- **Only exact establishes the same task.** Every hit carries `semanticEquivalence`:
+ `"identical"` for an exact hit, `"unverified"` for near and adapt. It also carries
+ `requiresReview`, `true` for every non-exact hit. A near hit is a similar task that no check
+ has shown to mean the same thing: review it against yours before reusing it.
+ `revalidation` answers a different question, namely whether the artifact and its
+ dependencies still hold.
+- **Near must also agree on behaviour.** A similar spec is offered as `near` only when the
two specs agree, in the same order, on these:
- operators and symbols, with their operands (`x + 1` vs `x - 1`, `a - b` vs `b - a`);
- numbers, identifiers, paths and negation;
+ - what each polarity, negation or direction word applies to. "Allow admins and deny
+ guests" vs "Deny admins and allow guests", or "from staging to production" vs "from
+ production to staging", conflict on `binding`, and so does the same set of words in a
+ different order;
- literals, typographic and backtick quotes included (“a b”, ``a b``);
- every whitespace run other than one space (line breaks, indentation, tabs, CRLF, columns);
- spelling (`parse` vs `Parse`, fullwidth or Cyrillic look-alikes) and invisible format
@@ -1203,6 +1226,7 @@ $ forge reuse query "debounce user input before firing search"
sim: minhash
NEAR hit (similarity 0.87) — module at src/lib/debounce.js
claim 9c41d2ab77e0 — `forge ledger blame 9c41d2ab` for its proof
+ near tier: a similar task, equivalence unverified — review it against yours before reusing
$ forge reuse query "quantum blockchain"
sim: minhash
diff --git a/docs/plans/substrate-v2/03-reuse-cache.md b/docs/plans/substrate-v2/03-reuse-cache.md
index f9c73d8..5fe0e4a 100644
--- a/docs/plans/substrate-v2/03-reuse-cache.md
+++ b/docs/plans/substrate-v2/03-reuse-cache.md
@@ -48,7 +48,12 @@ neighbourhood" are different questions:
exact tier now compares `keyHash`, a digest of the text's code units, and nothing is
normalized at that boundary; similarity search stays layout-blind. The near tier compares
the stored `key` with the semantic guard, so it requires a key of the current version
- that the ledger stored verbatim (`keyVerbatim`: no CRLF or non-NFC text to fold).
+ that the ledger stored verbatim (`keyVerbatim`: no CRLF or non-NFC text to fold). The
+ guard's `binding` kind (review Q01) also compares what each polarity or direction word
+ applies to, so "Allow admins and deny guests" and "Deny admins and allow guests", which
+ share every token, are held at adapt. No token check establishes that two texts mean the
+ same, so every hit states what it establishes: `semanticEquivalence` is `"identical"` only
+ for exact and `"unverified"` for near and adapt, which also carry `requiresReview: true`.
- **shape** (`spec`): identity plus typed placeholders for identifiers, paths, numbers and
string literals (`⟨ident⟩`, `⟨path⟩`, `⟨num⟩`, `⟨str⟩`).
diff --git a/docs/status/README.md b/docs/status/README.md
index 73f365e..135a3b6 100644
--- a/docs/status/README.md
+++ b/docs/status/README.md
@@ -34,8 +34,8 @@ The registry exists because the project's own rule — *do not assert without ev
place where every assertion's evidence can be looked up.
-48 claims — implemented 14 · measured 10 · partial 5 · reported 3 · hypothesis 6 · refuted 10.
-Assessed against commits `d2abfa69fb77`, `7eef61179d16`, `2699aa02397e`, `a6be83cac57b` (as of 2026-09-27).
+49 claims — implemented 15 · measured 10 · partial 5 · reported 3 · hypothesis 6 · refuted 10.
+Assessed against commits `d2abfa69fb77`, `7eef61179d16`, `2699aa02397e`, `a6be83cac57b`, `cf890405dc9f` (as of 2026-09-27).
| ID | Status | Assessed | Claim | Component · version | Evidence | Notes |
| --- | --- | --- | --- | --- | --- | --- |
@@ -59,22 +59,23 @@ Assessed against commits `d2abfa69fb77`, `7eef61179d16`, `2699aa02397e`, `a6be83
| `universal-router-target-guarantee` | **refuted** | 1.5.0 · `7eef6117` | The target:p objective guarantees a success rate of at least p. | Universal router (src/router, forge route universal) · 1.5.0 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist. Unchanged at 7eef611. |
| `universal-router-budget-contract` | **partial** | 1.7.3 · `2699aa02` | The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists. | Universal router (src/router, forge route universal) · 1.5.0 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`src/router/policy.js`](../../src/router/policy.js) | Review counterexamples: 2026-09-26 F12 fixed; 2026-09-27 suggestion 1 fixed. B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). The explicit half holds (re-assessed on 2699aa0): an infeasible budget returns ok:false, feasible:false, budgetMet:false, minimumExpectedCost and a reason (the CLI prints INFEASIBLE), with the least-bad cascade only as a labeled fallback. Every recommendation reports estimatedCostIfAllAttemptsRun, the sum of each attempt's expected cost; it was called maxPossibleCost until the 2026-09-27 review pointed out it is a modeled figure, not a bound on billed cost. |
| `universal-router-cascade-cost-independence` | **refuted** | 1.5.0 · `7eef6117` | A later cascade attempt costs the same in expectation whether or not earlier attempts failed. | Universal router cost model (src/router/policy.js) · 1.5.0 | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | The cascade cost formula assumes E[cost_i \| earlier attempts failed, x] = E[cost_i \| x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes. Unchanged at 7eef611. |
-| `route-outcome-provenance` | **partial** | 1.7.3 · `a6be83ca` | Outcomes recorded with forge route outcome are verified results. | Universal router outcome log (src/router/index.js) · derived provenance, verifier event contract v2 | [`src/router/index.js`](../../src/router/index.js)
[`test/router_universal.test.js`](../../test/router_universal.test.js)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | Review counterexamples: 2026-09-26 A01 fixed; 2026-09-26 A04 fixed; 2026-09-27 N06 fixed; 2026-09-27 N07 fixed. Re-assessed on a6be83c: N06 (with no evidence key, an unsigned verifier event earned verify-event) and N07 (an edited provenance label was read as verified) are regression and seeded property tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: an events file copied from another checkout, an old PASS backing a later attempt on other code, a legacy row's content key aliasing an attempt id, and a non-finite number in a signed event. Provenance is derived on every read from verifier events whose MAC verifies under this machine's key, that name this checkout, that finished on the code the outcome was recorded against, whose pass/fail verdict matches, and that back no other attempt; the stored label is informational. The event vouches for the verdict only: model and cost stay self-reported. Without --verify-run a row is self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable). |
+| `route-outcome-provenance` | **partial** | 1.7.3 · `a6be83ca` | Outcomes recorded with forge route outcome are verified results. | Universal router outcome log (src/router/index.js) · derived provenance, verifier event contract v2 | [`src/router/index.js`](../../src/router/index.js)
[`test/router_universal.test.js`](../../test/router_universal.test.js)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | Review counterexamples: 2026-09-26 A01 fixed; 2026-09-26 A04 fixed; 2026-09-27 N06 fixed; 2026-09-27 N07 fixed. Re-assessed on a6be83c: N06 (with no evidence key, an unsigned verifier event earned verify-event) and N07 (an edited provenance label was read as verified) are regression and seeded property tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: an events file copied from another checkout, an old PASS backing a later attempt on other code, a legacy row's content key aliasing an attempt id, and a non-finite number in a signed event. Provenance is derived on every read from verifier events whose MAC verifies under this machine's key, that name this checkout, that finished on the code the outcome was recorded against, whose pass/fail verdict matches, and that back no other attempt; the stored label is informational. The event vouches for the verdict only: model and cost stay self-reported. Without --verify-run a row is self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable). The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N06 and N07 fixtures closed. |
| `cost-reduction-90-target` | **hypothesis** | 1.4.3 · `d2abfa69` | Forgekit reduces coding-agent cost by about 90%. | Cost model (docs/plans/substrate-v2/05-cost-model.md) · plan, corrected 2026-09-26 | [`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`reports/cost-eval.md`](../../reports/cost-eval.md) | The owner's target. The 90.2 / 85.6 / 74.3% scenarios were derived from the refuted 0.62 routing factor and were withdrawn on 2026-09-26; separately estimated stage savings do not multiply into a total. No end-to-end cost has been measured. |
| `phase-p0-specs` | **implemented** | 1.4.3 · `d2abfa69` | P0: specs 00-08, ADR-0005 and ADR-0006 are merged. | Substrate v2 plan · v0.5.0 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/adr/0005-allow-runtime-dependencies.md`](../../docs/adr/0005-allow-runtime-dependencies.md)
[`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md) | |
| `phase-p1-ledger-core` | **implemented** | 1.4.3 · `d2abfa69` | P1: the claim ledger stores content-addressed claims whose confidence moves only with independent oracle evidence. | Ledger (src/ledger.js, src/ledger_store.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger.js`](../../src/ledger.js)
[`test/ledger.test.js`](../../test/ledger.test.js) | The review found evidence-semantics defects at the assessed commit: aliases of one git object counted as independent evidence (F06), and a rewritten lesson inherited the old statement's confidence (F07); aedddf5 (committed after the assessed commit) repairs both with regression tests. See ledger-evidence-independence. |
| `phase-p2-team-sync` | **implemented** | 1.4.3 · `d2abfa69` | P2: ledgers merge across teammates by git union, converging regardless of order. | Ledger sync (src/ledger_sync.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger_sync.js`](../../src/ledger_sync.js)
[`test/ledger_sync.test.js`](../../test/ledger_sync.test.js) | Storage convergence, property-tested; not semantic agreement or factual correctness. The read-path flip (merged legacy + ledger view) and ledger-only writes shipped after P2. |
-| `phase-p3-reuse-cache` | **implemented** | 1.7.3 · `a6be83ca` | P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one. | Reuse cache (src/reuse.js, forge reuse) · reuse key v3, dependency contracts v2 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Review counterexamples: 2026-09-26 F04 fixed; 2026-09-26 F05 fixed; 2026-09-27 N01 fixed; 2026-09-27 N08 fixed. Re-assessed on a6be83c: the acceptance counterexamples found at d2abfa6 (F04, F05), at 1.7.1 (N01 exact identity collapsing whitespace inside literals; N08 destructured dependency contracts) and by an internal adversarial re-review of 2699aa0 (near-tier guard gaps, missed import forms, export aliases) are repaired with regression and seeded property tests. |
-| `phase-p4-context-assembly` | **partial** | 1.7.3 · `a6be83ca` | P4: context assembly never exceeds its token budget and reports a computed missing set. | Context assembly (src/context.js, forge context) · context assembly with definition extents (atlas v5) | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/context.js`](../../src/context.js) | Review counterexamples: 2026-09-26 F02 fixed; 2026-09-26 F03 fixed; 2026-09-27 N04 fixed. Re-assessed on a6be83c: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, a pointer is not coverage (F02, F03), and a definition is delivered only whole (N04, and the extent counterexamples of an internal adversarial re-review of 2699aa0). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. |
+| `phase-p3-reuse-cache` | **implemented** | 1.7.3 · `a6be83ca` | P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one. | Reuse cache (src/reuse.js, forge reuse) · reuse key v3, dependency contracts v2 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Review counterexamples: 2026-09-26 F04 fixed; 2026-09-26 F05 fixed; 2026-09-27 N01 fixed; 2026-09-27 N08 fixed. Re-assessed on a6be83c: the acceptance counterexamples found at d2abfa6 (F04, F05), at 1.7.1 (N01 exact identity collapsing whitespace inside literals; N08 destructured dependency contracts) and by an internal adversarial re-review of 2699aa0 (near-tier guard gaps, missed import forms, export aliases) are repaired with regression and seeded property tests. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N01 and N08 fixtures closed. |
+| `phase-p4-context-assembly` | **partial** | 1.7.3 · `a6be83ca` | P4: context assembly never exceeds its token budget and reports a computed missing set. | Context assembly (src/context.js, forge context) · context assembly with definition extents (atlas v5) | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/context.js`](../../src/context.js) | Review counterexamples: 2026-09-26 F02 fixed; 2026-09-26 F03 fixed; 2026-09-27 N04 fixed. Re-assessed on a6be83c: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, a pointer is not coverage (F02, F03), and a definition is delivered only whole (N04, and the extent counterexamples of an internal adversarial re-review of 2699aa0). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N04 fixture closed. |
| `phase-p5-loop-closure` | **implemented** | 1.4.3 · `d2abfa69` | P5: outcomes move the confidence of the claims that informed an action; repeated failures mint a diagnosis; imagine dry-runs the selected tests. | Loop closure (src/diagnose.js, src/imagine.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/imagine.js`](../../src/imagine.js)
[`test/imagine.test.js`](../../test/imagine.test.js) | imagine --run is an isolated git checkout of the committed baseline, not a security sandbox, and runs selected files under node --test. |
| `phase-p6-ui-quality-gate` | **implemented** | 1.4.3 · `d2abfa69` | P6: forge uicheck flags a known-template fixture and passes a project-conformant one, with no LLM calls. | UI checks (src/uicheck.js, forge uicheck) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/plans/substrate-v2/07-ui-quality-gate.md`](../../docs/plans/substrate-v2/07-ui-quality-gate.md)
[`test/uicheck.test.js`](../../test/uicheck.test.js) | The checks are advisory measurements (contrast, token conformance, template distance, fingerprint similarity); none measures accessibility or user value, and none is a blocking hook. |
| `phase-p7-dashboard` | **implemented** | 1.4.3 · `d2abfa69` | P7: forge dash renders the ledger, cost meter, cache rate and blast radius offline. | Dashboard (src/dash.js, forge dash) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/dash.js`](../../src/dash.js)
[`test/dash.test.js`](../../test/dash.test.js) | The review found read routes answering a foreign Host header at the assessed commit (F13), repaired in aedddf5 (committed after the assessed commit). Cost panels show stage self-estimates, not end-to-end spend. |
| `phase-p8-evaluation` | **partial** | 1.4.3 · `d2abfa69` | P8: a measured, not asserted, end-to-end cost figure per stage is published in reports/. | Cost evaluation (src/cost_report.js, forge cost --stages) · 1.4.3 | [`reports/cost-eval.md`](../../reports/cost-eval.md)
[`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md) | Stage instrumentation and the stage report exist; no paired end-to-end run has been measured and reports/cost-eval.md holds no data. |
-| `context-completeness` | **implemented** | 1.7.3 · `a6be83ca` | forge context COMPLETE means every required item was delivered as content within the (estimated) token budget, every required definition whole. | Context assembly (src/context.js, forge context) · context assembly with definition extents (atlas v5) | [`src/context.js`](../../src/context.js)
[`test/context.test.js`](../../test/context.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md) | Review counterexamples: 2026-09-26 F02 fixed; 2026-09-26 F03 fixed; 2026-09-27 N04 fixed. Re-assessed on a6be83c: N04 (lines 1-41 of a 103-line function counted as a delivered definition) is a regression test, and so are the extent counterexamples an internal adversarial re-review of 2699aa0 found (generic constraints, return-type literals, overloads, Go result types, template constants, Ruby classes, Python strings and continuations, JSX apostrophes, preprocessor branches, a stale atlas). A span or head covers a definition only when it shows the whole extent the atlas records and trusts, and a cut one stays pending and is listed in `partial`. A definition with no trusted extent, or in a file edited since the atlas was built, is delivered only by the whole file. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted. |
-| `verify-pass-binding` | **implemented** | 1.7.3 · `a6be83ca` | A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite. | Verification (src/verify.js, forge verify) · fingerprint manifest-v3, verifier event contract v2 | [`src/verify.js`](../../src/verify.js)
[`test/verify.test.js`](../../test/verify.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Review counterexamples: 2026-09-26 F01 fixed; 2026-09-26 F08 fixed; 2026-09-26 F10 fixed; 2026-09-27 N03 fixed; 2026-09-27 N05 fixed. Re-assessed on a6be83c: N03 (Node's `-r` preload flag read as a recursive workspace run, so a failing workspace never ran) and N05 (a test rewrote code it imported from an untracked nested repository without changing the fingerprint) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: unknown options, shell builtins, narrowing configuration and environment, replayed tool caches, negation shapes, yarn and nx member rules, Bun's own runner, gitlinks without .gitmodules, a nested repository's own config and excludes, and a script-shell that is not a shell. A root script covers the workspaces only when it is a recognized, unfiltered, uncached recursive run whose failure reaches its exit status and nothing narrows it; each package's verdict is labelled measured, inferred or declared. Nested repositories, submodules and gitlinks are bound by their own code state under the outer config; code that still cannot be bound makes a PASS INCOMPLETE unless declared in verify.external, which every verifier event records. Scope: gitignored files, declared verify.generated outputs and interpreter caches are not bound; forge trusts the repository's own tooling (a replaced test runner or tool binary is outside the check); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted. |
+| `context-completeness` | **implemented** | 1.7.3 · `a6be83ca` | forge context COMPLETE means every required item was delivered as content within the (estimated) token budget, every required definition whole. | Context assembly (src/context.js, forge context) · context assembly with definition extents (atlas v5) | [`src/context.js`](../../src/context.js)
[`test/context.test.js`](../../test/context.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md) | Review counterexamples: 2026-09-26 F02 fixed; 2026-09-26 F03 fixed; 2026-09-27 N04 fixed. Re-assessed on a6be83c: N04 (lines 1-41 of a 103-line function counted as a delivered definition) is a regression test, and so are the extent counterexamples an internal adversarial re-review of 2699aa0 found (generic constraints, return-type literals, overloads, Go result types, template constants, Ruby classes, Python strings and continuations, JSX apostrophes, preprocessor branches, a stale atlas). A span or head covers a definition only when it shows the whole extent the atlas records and trusts, and a cut one stays pending and is listed in `partial`. A definition with no trusted extent, or in a file edited since the atlas was built, is delivered only by the whole file. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N04 fixture closed. |
+| `verify-pass-binding` | **implemented** | unreleased · `cf890405` | A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite. | Verification (src/verify.js, forge verify) · fingerprint manifest-v3, verifier event contract v2 | [`src/verify.js`](../../src/verify.js)
[`test/verify.test.js`](../../test/verify.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Review counterexamples: 2026-09-26 F01 fixed; 2026-09-26 F08 fixed; 2026-09-26 F10 fixed; 2026-09-27 N03 fixed; 2026-09-27 N05 fixed; 2026-09-27 Q02 fixed. Re-assessed on cf89040: Q02 (a root `./tools/npm test --workspaces` credited as npm's recursive run from the file's basename, so a stub that exited 0 hid a failing workspace) is a regression test with the review's fixture, which now runs the member and FAILs. A tool is recognized only by its bare name or its own node_modules/.bin install; command-line variables are allowlisted (PATH and preloads refused); a package-manager or system-program shim on the script PATH, another package's binary or an out-of-package link under a tool's name, a same-named Windows executable in the root, a yarnPath outside .yarn/releases or a packageManager URL refuses the run; a seeded property test covers tools named through paths and PATH overrides. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N03 and N05 fixtures closed. Earlier, on a6be83c: N03 (Node's `-r` preload flag read as a recursive workspace run, so a failing workspace never ran) and N05 (a test rewrote code it imported from an untracked nested repository without changing the fingerprint) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: unknown options, shell builtins, narrowing configuration and environment, replayed tool caches, negation shapes, yarn and nx member rules, Bun's own runner, gitlinks without .gitmodules, a nested repository's own config and excludes, and a script-shell that is not a shell. A root script covers the workspaces only when it is a recognized, unfiltered, uncached recursive run whose failure reaches its exit status and nothing narrows it; each package's verdict is labelled measured, inferred or declared. Nested repositories, submodules and gitlinks are bound by their own code state under the outer config; code that still cannot be bound makes a PASS INCOMPLETE unless declared in verify.external, which every verifier event records. Scope: gitignored files, declared verify.generated outputs and interpreter caches are not bound; beyond the identity and shadow checks forge trusts the installed tools themselves (what a genuine package's binary or the test runner does is outside the check); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted. |
| `ledger-evidence-independence` | **partial** | 1.5.0 · `7eef6117` | A claim's confidence rises only with independent evidence for that claim. | Ledger (src/ledger_store.js, src/ledger_bridge.js) · 1.5.0 | [`src/ledger.js`](../../src/ledger.js)
[`src/ledger_store.js`](../../src/ledger_store.js)
[`src/ledger_bridge.js`](../../src/ledger_bridge.js)
[`test/ledger_store.test.js`](../../test/ledger_store.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: abbreviations of one git object count as one event, a replayed or re-cited ref cannot refresh decay, and a non-equivalent rewrite no longer inherits trust (F06, F07 regression tests; a seeded property test over spelling mixes, replays and order). Not yet held: two different references to the same underlying run (for example a CI run id and its commit) still count as two events. At d2abfa6 this claim was refuted. |
-| `similar-rules-never-merged` | **implemented** | 1.7.3 · `a6be83ca` | Lesson consolidation and ledger compaction never merge, archive or drop a rule because it is similar to another: only the same statement is merged, and only a refuted claim with exactly a lesson's text drops it. | Consolidation and compaction (src/learn_consolidate.js, src/ledger_retention.js) · exact-statement merging | [`src/learn_consolidate.js`](../../src/learn_consolidate.js)
[`src/ledger_retention.js`](../../src/ledger_retention.js)
[`src/semantic_guard.js`](../../src/semantic_guard.js)
[`test/learn_consolidate.test.js`](../../test/learn_consolidate.test.js)
[`test/ledger_retention.test.js`](../../test/ledger_retention.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Review counterexamples: 2026-09-26 F16 fixed; 2026-09-27 N02 fixed. Re-assessed on a6be83c: N02 ("Allow admins and deny guests" and "Deny admins and allow guests" with a shared tail merged into one rule, the token-level guard seeing no conflict) is a regression test, with subject, direction, number and negation-scope swaps, and a seeded property test over role and value swaps; so are the counterexamples an internal adversarial re-review of 2699aa0 found (statement keys that folded `./...` into `./`, `save!` into `save` and inner whitespace; facts with different names and lessons with different triggers archived as duplicates; one repo's retraction outvoting a live copy). Near-duplicates are kept and listed as proposed; similar pairs that differ in a behaviour-bearing token are listed as conflicts; a lesson resembling a refuted claim is kept and flagged. A person merges what similarity proposes. Scope: the same statement is the same text up to its edge whitespace and one sentence period after a word, with the same fact name, lesson trigger and scope. |
-| `reuse-exact-identity` | **implemented** | 1.7.3 · `a6be83ca` | An exact reuse hit serves an artifact minted for the byte-identical specification, and only while its bytes and dependency contracts still match. | Reuse cache (src/reuse.js) · reuse key v3, dependency contracts v2 | [`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Review counterexamples: 2026-09-26 F04 fixed; 2026-09-26 F05 fixed; 2026-09-27 N01 fixed; 2026-09-27 N08 fixed. Re-assessed on a6be83c: N01 (a spec differing only in whitespace inside a literal exact-hit, the key having collapsed whitespace and folded NFC) and N08 (a destructured-parameter change left the dependency contract unchanged, and same-name symbols resolved by file order) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: near hits on older or normalized keys; operator, order, whitespace, spelling and typographic-literal differences the guard missed; dependencies missed or mis-bound through default, namespace, require, dynamic, aliased and re-exported imports. Seeded property tests cover quoted whitespace, code points and destructured parameters. The exact tier compares a digest of the spec's code units; near needs a current, verbatim-stored key and the semantic guard; a named import's contract hashes its defining declaration minus the body, and any other import records the module's digest. Scope: inline code the ledger would rewrite (CRLF, non-NFC) is refused at mint; contracts need a fresh atlas with definition extents, otherwise the hit is marked requiresRevalidation; a class's or object literal's contract includes its method bodies (conservative). At d2abfa6 this claim was refuted. |
+| `similar-rules-never-merged` | **implemented** | unreleased · `cf890405` | Lesson consolidation and ledger compaction never merge, archive or drop a rule because it is similar to another: only the same statement is merged, and only a refuted claim with exactly a lesson's text drops it. | Consolidation and compaction (src/learn_consolidate.js, src/ledger_retention.js) · exact-statement merging | [`src/learn_consolidate.js`](../../src/learn_consolidate.js)
[`src/ledger_retention.js`](../../src/ledger_retention.js)
[`src/semantic_guard.js`](../../src/semantic_guard.js)
[`test/learn_consolidate.test.js`](../../test/learn_consolidate.test.js)
[`test/ledger_retention.test.js`](../../test/ledger_retention.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`test/semantic_guard.test.js`](../../test/semantic_guard.test.js) | Review counterexamples: 2026-09-26 F16 fixed; 2026-09-27 N02 fixed. Re-assessed on cf89040, after the semantic guard gained its binding kind (review Q01). The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N02 fixture closed: two rules kept, none merged. Earlier, on a6be83c: N02 ("Allow admins and deny guests" and "Deny admins and allow guests" with a shared tail merged into one rule, the token-level guard seeing no conflict) is a regression test, with subject, direction, number and negation-scope swaps, and a seeded property test over role and value swaps; so are the counterexamples an internal adversarial re-review of 2699aa0 found (statement keys that folded `./...` into `./`, `save!` into `save` and inner whitespace; facts with different names and lessons with different triggers archived as duplicates; one repo's retraction outvoting a live copy). Near-duplicates are kept and listed as proposed; similar pairs that differ in a behaviour-bearing token, or in what a polarity or direction word applies to (the N02 pair itself, since cf89040), are listed as conflicts; a lesson resembling a refuted claim is kept and flagged. A person merges what similarity proposes. Scope: the same statement is the same text up to its edge whitespace and one sentence period after a word, with the same fact name, lesson trigger and scope. |
+| `reuse-exact-identity` | **implemented** | 1.7.3 · `a6be83ca` | An exact reuse hit serves an artifact minted for the byte-identical specification, and only while its bytes and dependency contracts still match. | Reuse cache (src/reuse.js) · reuse key v3, dependency contracts v2 | [`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Review counterexamples: 2026-09-26 F04 fixed; 2026-09-26 F05 fixed; 2026-09-27 N01 fixed; 2026-09-27 N08 fixed. Re-assessed on a6be83c: N01 (a spec differing only in whitespace inside a literal exact-hit, the key having collapsed whitespace and folded NFC) and N08 (a destructured-parameter change left the dependency contract unchanged, and same-name symbols resolved by file order) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: near hits on older or normalized keys; operator, order, whitespace, spelling and typographic-literal differences the guard missed; dependencies missed or mis-bound through default, namespace, require, dynamic, aliased and re-exported imports. Seeded property tests cover quoted whitespace, code points and destructured parameters. The exact tier compares a digest of the spec's code units; near needs a current, verbatim-stored key and the semantic guard; a named import's contract hashes its defining declaration minus the body, and any other import records the module's digest. Scope: inline code the ledger would rewrite (CRLF, non-NFC) is refused at mint; contracts need a fresh atlas with definition extents, otherwise the hit is marked requiresRevalidation; a class's or object literal's contract includes its method bodies (conservative). At d2abfa6 this claim was refuted. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N01 and N08 fixtures closed. |
+| `reuse-review-contract` | **implemented** | unreleased · `cf890405` | A non-exact reuse hit is never presented as the same task: it carries semanticEquivalence "unverified" and requiresReview true, and a known relational reversal (a subject bound to the opposite permission, a swapped source and destination, the same words rearranged) is held at adapt. | Reuse cache (src/reuse.js) and semantic guard (src/semantic_guard.js) · reuse key v3, guard binding kind | [`src/reuse.js`](../../src/reuse.js)
[`src/semantic_guard.js`](../../src/semantic_guard.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/semantic_guard.test.js`](../../test/semantic_guard.test.js) | Review counterexamples: 2026-09-27 Q01 fixed. Q01: "Deny admins and allow guests…" near-hit an artifact verified for "Allow admins and deny guests…" (MinHash 0.828, no guard conflict) and was labelled a reworded match. It is a regression test, and is now held at adapt with the binding conflict named; so are source/destination, into, negation-scope, quoted-literal and possessive role swaps. Every hit states what it establishes: semanticEquivalence is "identical" only for exact (the byte-identical spec) and "unverified" for near and adapt, which carry requiresReview: true in forge reuse query --json and the gate. Scope: the binding check is a token heuristic (each polarity, negation or direction word binds the next content word of its clause, and the same words in another order conflict); it holds known reversal shapes back from near but cannot establish that two texts mean the same, which is why no non-exact hit is presented as equivalent. |
| `imagine-sandbox` | **refuted** | 1.5.0 · `7eef6117` | forge imagine --run executes the selected tests in a sandbox. | Consequence simulation (src/imagine.js, forge imagine) · 1.5.0 | [`src/imagine.js`](../../src/imagine.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Still false at 7eef611, and no longer claimed: it is an isolated git checkout (a detached-HEAD worktree) that isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. The CLI and docs now say "an isolated checkout of HEAD, not a security sandbox", and a suite that is not node:test gets an explicit unsupported-runner result. |
| `integrations-emission` | **implemented** | 1.4.3 · `d2abfa69` | Forgekit emits native config for ten coding tools plus MCP config for Roo Code and VS Code. | Config compiler (src/sync.js, src/emit) · 1.4.3 | [`docs/INTEGRATIONS.md`](../../docs/INTEGRATIONS.md)
[`test/sync.test.js`](../../test/sync.test.js)
[`test/mcp.test.js`](../../test/mcp.test.js) | Emission is tested; no host tool is launched in CI. Automatic hooks and in-agent enforcement exist only on Claude Code; a git pre-commit gate covers any tool that commits through git. |
| `pcm-evidence-referenced-memory` | **implemented** | 1.4.3 · `d2abfa69` | Proof-carrying memory stores content-addressed claims that carry references to their evidence. | Ledger (src/ledger.js) · 1.4.3 | [`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md)
[`src/ledger.js`](../../src/ledger.js) | A name, not a formal proof: there is no theorem prover in the loop. |
diff --git a/docs/status/claims.json b/docs/status/claims.json
index 541b005..ba6b017 100644
--- a/docs/status/claims.json
+++ b/docs/status/claims.json
@@ -328,7 +328,7 @@
"test/router_universal.test.js",
"docs/UNIVERSAL_ROUTING.md"
],
- "notes": "Re-assessed on a6be83c: N06 (with no evidence key, an unsigned verifier event earned verify-event) and N07 (an edited provenance label was read as verified) are regression and seeded property tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: an events file copied from another checkout, an old PASS backing a later attempt on other code, a legacy row's content key aliasing an attempt id, and a non-finite number in a signed event. Provenance is derived on every read from verifier events whose MAC verifies under this machine's key, that name this checkout, that finished on the code the outcome was recorded against, whose pass/fail verdict matches, and that back no other attempt; the stored label is informational. The event vouches for the verdict only: model and cost stay self-reported. Without --verify-run a row is self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable).",
+ "notes": "Re-assessed on a6be83c: N06 (with no evidence key, an unsigned verifier event earned verify-event) and N07 (an edited provenance label was read as verified) are regression and seeded property tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: an events file copied from another checkout, an old PASS backing a later attempt on other code, a legacy row's content key aliasing an attempt id, and a non-finite number in a signed event. Provenance is derived on every read from verifier events whose MAC verifies under this machine's key, that name this checkout, that finished on the code the outcome was recorded against, whose pass/fail verdict matches, and that back no other attempt; the stored label is informational. The event vouches for the verdict only: model and cost stay self-reported. Without --verify-run a row is self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable). The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N06 and N07 fixtures closed.",
"counterexamples": [
{
"review": "2026-09-26",
@@ -425,7 +425,7 @@
"test/reuse.test.js",
"test/trust_properties.test.js"
],
- "notes": "Re-assessed on a6be83c: the acceptance counterexamples found at d2abfa6 (F04, F05), at 1.7.1 (N01 exact identity collapsing whitespace inside literals; N08 destructured dependency contracts) and by an internal adversarial re-review of 2699aa0 (near-tier guard gaps, missed import forms, export aliases) are repaired with regression and seeded property tests.",
+ "notes": "Re-assessed on a6be83c: the acceptance counterexamples found at d2abfa6 (F04, F05), at 1.7.1 (N01 exact identity collapsing whitespace inside literals; N08 destructured dependency contracts) and by an internal adversarial re-review of 2699aa0 (near-tier guard gaps, missed import forms, export aliases) are repaired with regression and seeded property tests. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N01 and N08 fixtures closed.",
"counterexamples": [
{
"review": "2026-09-26",
@@ -462,7 +462,7 @@
"docs/plans/substrate-v2/00-overview.md",
"src/context.js"
],
- "notes": "Re-assessed on a6be83c: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, a pointer is not coverage (F02, F03), and a definition is delivered only whole (N04, and the extent counterexamples of an internal adversarial re-review of 2699aa0). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered.",
+ "notes": "Re-assessed on a6be83c: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, a pointer is not coverage (F02, F03), and a definition is delivered only whole (N04, and the extent counterexamples of an internal adversarial re-review of 2699aa0). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N04 fixture closed.",
"counterexamples": [
{
"review": "2026-09-26",
@@ -555,7 +555,7 @@
"test/trust_properties.test.js",
"docs/plans/substrate-v2/04-context-assembly.md"
],
- "notes": "Re-assessed on a6be83c: N04 (lines 1-41 of a 103-line function counted as a delivered definition) is a regression test, and so are the extent counterexamples an internal adversarial re-review of 2699aa0 found (generic constraints, return-type literals, overloads, Go result types, template constants, Ruby classes, Python strings and continuations, JSX apostrophes, preprocessor branches, a stale atlas). A span or head covers a definition only when it shows the whole extent the atlas records and trusts, and a cut one stays pending and is listed in `partial`. A definition with no trusted extent, or in a file edited since the atlas was built, is delivered only by the whole file. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted.",
+ "notes": "Re-assessed on a6be83c: N04 (lines 1-41 of a 103-line function counted as a delivered definition) is a regression test, and so are the extent counterexamples an internal adversarial re-review of 2699aa0 found (generic constraints, return-type literals, overloads, Go result types, template constants, Ruby classes, Python strings and continuations, JSX apostrophes, preprocessor branches, a stale atlas). A span or head covers a definition only when it shows the whole extent the atlas records and trusts, and a cut one stays pending and is listed in `partial`. A definition with no trusted extent, or in a file edited since the atlas was built, is delivered only by the whole file. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N04 fixture closed.",
"counterexamples": [
{
"review": "2026-09-26",
@@ -579,8 +579,8 @@
"claim": "A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite.",
"component": "Verification (src/verify.js, forge verify)",
"version": "fingerprint manifest-v3, verifier event contract v2",
- "source_commit": "a6be83cac57b8b728c6a9b2e14430f2b7186c20b",
- "assessed_release": "1.7.3",
+ "source_commit": "cf890405dc9f45b99cc87304efc857ef9921addd",
+ "assessed_release": "unreleased",
"status": "implemented",
"evidence": [
"src/verify.js",
@@ -588,7 +588,7 @@
"test/trust_properties.test.js",
"docs/GUIDE.md"
],
- "notes": "Re-assessed on a6be83c: N03 (Node's `-r` preload flag read as a recursive workspace run, so a failing workspace never ran) and N05 (a test rewrote code it imported from an untracked nested repository without changing the fingerprint) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: unknown options, shell builtins, narrowing configuration and environment, replayed tool caches, negation shapes, yarn and nx member rules, Bun's own runner, gitlinks without .gitmodules, a nested repository's own config and excludes, and a script-shell that is not a shell. A root script covers the workspaces only when it is a recognized, unfiltered, uncached recursive run whose failure reaches its exit status and nothing narrows it; each package's verdict is labelled measured, inferred or declared. Nested repositories, submodules and gitlinks are bound by their own code state under the outer config; code that still cannot be bound makes a PASS INCOMPLETE unless declared in verify.external, which every verifier event records. Scope: gitignored files, declared verify.generated outputs and interpreter caches are not bound; forge trusts the repository's own tooling (a replaced test runner or tool binary is outside the check); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted.",
+ "notes": "Re-assessed on cf89040: Q02 (a root `./tools/npm test --workspaces` credited as npm's recursive run from the file's basename, so a stub that exited 0 hid a failing workspace) is a regression test with the review's fixture, which now runs the member and FAILs. A tool is recognized only by its bare name or its own node_modules/.bin install; command-line variables are allowlisted (PATH and preloads refused); a package-manager or system-program shim on the script PATH, another package's binary or an out-of-package link under a tool's name, a same-named Windows executable in the root, a yarnPath outside .yarn/releases or a packageManager URL refuses the run; a seeded property test covers tools named through paths and PATH overrides. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N03 and N05 fixtures closed. Earlier, on a6be83c: N03 (Node's `-r` preload flag read as a recursive workspace run, so a failing workspace never ran) and N05 (a test rewrote code it imported from an untracked nested repository without changing the fingerprint) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: unknown options, shell builtins, narrowing configuration and environment, replayed tool caches, negation shapes, yarn and nx member rules, Bun's own runner, gitlinks without .gitmodules, a nested repository's own config and excludes, and a script-shell that is not a shell. A root script covers the workspaces only when it is a recognized, unfiltered, uncached recursive run whose failure reaches its exit status and nothing narrows it; each package's verdict is labelled measured, inferred or declared. Nested repositories, submodules and gitlinks are bound by their own code state under the outer config; code that still cannot be bound makes a PASS INCOMPLETE unless declared in verify.external, which every verifier event records. Scope: gitignored files, declared verify.generated outputs and interpreter caches are not bound; beyond the identity and shadow checks forge trusts the installed tools themselves (what a genuine package's binary or the test runner does is outside the check); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted.",
"counterexamples": [
{
"review": "2026-09-26",
@@ -614,6 +614,11 @@
"review": "2026-09-27",
"id": "N05",
"resolution": "fixed"
+ },
+ {
+ "review": "2026-09-27",
+ "id": "Q02",
+ "resolution": "fixed"
}
]
},
@@ -639,8 +644,8 @@
"claim": "Lesson consolidation and ledger compaction never merge, archive or drop a rule because it is similar to another: only the same statement is merged, and only a refuted claim with exactly a lesson's text drops it.",
"component": "Consolidation and compaction (src/learn_consolidate.js, src/ledger_retention.js)",
"version": "exact-statement merging",
- "source_commit": "a6be83cac57b8b728c6a9b2e14430f2b7186c20b",
- "assessed_release": "1.7.3",
+ "source_commit": "cf890405dc9f45b99cc87304efc857ef9921addd",
+ "assessed_release": "unreleased",
"status": "implemented",
"evidence": [
"src/learn_consolidate.js",
@@ -648,7 +653,8 @@
"src/semantic_guard.js",
"test/learn_consolidate.test.js",
"test/ledger_retention.test.js",
- "test/trust_properties.test.js"
+ "test/trust_properties.test.js",
+ "test/semantic_guard.test.js"
],
"counterexamples": [
{
@@ -662,7 +668,7 @@
"resolution": "fixed"
}
],
- "notes": "Re-assessed on a6be83c: N02 (\"Allow admins and deny guests\" and \"Deny admins and allow guests\" with a shared tail merged into one rule, the token-level guard seeing no conflict) is a regression test, with subject, direction, number and negation-scope swaps, and a seeded property test over role and value swaps; so are the counterexamples an internal adversarial re-review of 2699aa0 found (statement keys that folded `./...` into `./`, `save!` into `save` and inner whitespace; facts with different names and lessons with different triggers archived as duplicates; one repo's retraction outvoting a live copy). Near-duplicates are kept and listed as proposed; similar pairs that differ in a behaviour-bearing token are listed as conflicts; a lesson resembling a refuted claim is kept and flagged. A person merges what similarity proposes. Scope: the same statement is the same text up to its edge whitespace and one sentence period after a word, with the same fact name, lesson trigger and scope."
+ "notes": "Re-assessed on cf89040, after the semantic guard gained its binding kind (review Q01). The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N02 fixture closed: two rules kept, none merged. Earlier, on a6be83c: N02 (\"Allow admins and deny guests\" and \"Deny admins and allow guests\" with a shared tail merged into one rule, the token-level guard seeing no conflict) is a regression test, with subject, direction, number and negation-scope swaps, and a seeded property test over role and value swaps; so are the counterexamples an internal adversarial re-review of 2699aa0 found (statement keys that folded `./...` into `./`, `save!` into `save` and inner whitespace; facts with different names and lessons with different triggers archived as duplicates; one repo's retraction outvoting a live copy). Near-duplicates are kept and listed as proposed; similar pairs that differ in a behaviour-bearing token, or in what a polarity or direction word applies to (the N02 pair itself, since cf89040), are listed as conflicts; a lesson resembling a refuted claim is kept and flagged. A person merges what similarity proposes. Scope: the same statement is the same text up to its edge whitespace and one sentence period after a word, with the same fact name, lesson trigger and scope."
},
{
"id": "reuse-exact-identity",
@@ -677,7 +683,7 @@
"test/reuse.test.js",
"test/trust_properties.test.js"
],
- "notes": "Re-assessed on a6be83c: N01 (a spec differing only in whitespace inside a literal exact-hit, the key having collapsed whitespace and folded NFC) and N08 (a destructured-parameter change left the dependency contract unchanged, and same-name symbols resolved by file order) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: near hits on older or normalized keys; operator, order, whitespace, spelling and typographic-literal differences the guard missed; dependencies missed or mis-bound through default, namespace, require, dynamic, aliased and re-exported imports. Seeded property tests cover quoted whitespace, code points and destructured parameters. The exact tier compares a digest of the spec's code units; near needs a current, verbatim-stored key and the semantic guard; a named import's contract hashes its defining declaration minus the body, and any other import records the module's digest. Scope: inline code the ledger would rewrite (CRLF, non-NFC) is refused at mint; contracts need a fresh atlas with definition extents, otherwise the hit is marked requiresRevalidation; a class's or object literal's contract includes its method bodies (conservative). At d2abfa6 this claim was refuted.",
+ "notes": "Re-assessed on a6be83c: N01 (a spec differing only in whitespace inside a literal exact-hit, the key having collapsed whitespace and folded NFC) and N08 (a destructured-parameter change left the dependency contract unchanged, and same-name symbols resolved by file order) are regression tests, and so are the counterexamples an internal adversarial re-review of 2699aa0 found: near hits on older or normalized keys; operator, order, whitespace, spelling and typographic-literal differences the guard missed; dependencies missed or mis-bound through default, namespace, require, dynamic, aliased and re-exported imports. Seeded property tests cover quoted whitespace, code points and destructured parameters. The exact tier compares a digest of the spec's code units; near needs a current, verbatim-stored key and the semantic guard; a named import's contract hashes its defining declaration minus the body, and any other import records the module's digest. Scope: inline code the ledger would rewrite (CRLF, non-NFC) is refused at mint; contracts need a fresh atlas with definition extents, otherwise the hit is marked requiresRevalidation; a class's or object literal's contract includes its method bodies (conservative). At d2abfa6 this claim was refuted. The 2026-09-27 recheck of 1.7.3 (6d98121) independently confirmed the N01 and N08 fixtures closed.",
"counterexamples": [
{
"review": "2026-09-26",
@@ -701,6 +707,29 @@
}
]
},
+ {
+ "id": "reuse-review-contract",
+ "claim": "A non-exact reuse hit is never presented as the same task: it carries semanticEquivalence \"unverified\" and requiresReview true, and a known relational reversal (a subject bound to the opposite permission, a swapped source and destination, the same words rearranged) is held at adapt.",
+ "component": "Reuse cache (src/reuse.js) and semantic guard (src/semantic_guard.js)",
+ "version": "reuse key v3, guard binding kind",
+ "source_commit": "cf890405dc9f45b99cc87304efc857ef9921addd",
+ "assessed_release": "unreleased",
+ "status": "implemented",
+ "evidence": [
+ "src/reuse.js",
+ "src/semantic_guard.js",
+ "test/reuse.test.js",
+ "test/semantic_guard.test.js"
+ ],
+ "notes": "Q01: \"Deny admins and allow guests…\" near-hit an artifact verified for \"Allow admins and deny guests…\" (MinHash 0.828, no guard conflict) and was labelled a reworded match. It is a regression test, and is now held at adapt with the binding conflict named; so are source/destination, into, negation-scope, quoted-literal and possessive role swaps. Every hit states what it establishes: semanticEquivalence is \"identical\" only for exact (the byte-identical spec) and \"unverified\" for near and adapt, which carry requiresReview: true in forge reuse query --json and the gate. Scope: the binding check is a token heuristic (each polarity, negation or direction word binds the next content word of its clause, and the same words in another order conflict); it holds known reversal shapes back from near but cannot establish that two texts mean the same, which is why no non-exact hit is presented as equivalent.",
+ "counterexamples": [
+ {
+ "review": "2026-09-27",
+ "id": "Q01",
+ "resolution": "fixed"
+ }
+ ]
+ },
{
"id": "imagine-sandbox",
"claim": "forge imagine --run executes the selected tests in a sandbox.",
diff --git a/mintlify/changelog/overview.mdx b/mintlify/changelog/overview.mdx
index ec582b2..b89a381 100644
--- a/mintlify/changelog/overview.mdx
+++ b/mintlify/changelog/overview.mdx
@@ -18,6 +18,18 @@ This page is generated from `CHANGELOG.md` by `forge docs render`, and `forge do
CI when it falls behind, so it cannot drift from the release notes again.
{/* forge:render:changelog:begin (generated by `forge docs render` — do not edit) */}
+
+
+**Fixed**
+
+- **A local program named like a package manager no longer covers the workspaces (Q02).**
+- **A near reuse hit is no longer presented as the same task (Q01).**
+- **The claims check no longer fails in a checkout that lacks newer release tags.**
+
+[Full notes for Unreleased →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#unreleased)
+
+
+
**Fixed**
diff --git a/mintlify/cli/memory.mdx b/mintlify/cli/memory.mdx
index 80394f9..5e26b4d 100644
--- a/mintlify/cli/memory.mdx
+++ b/mintlify/cli/memory.mdx
@@ -92,12 +92,18 @@ forge reuse stats # cache stats
`"a b"`, indentation and Unicode code points all count, as do case, operators and
punctuation, so `age >= 18` and `age <= 18` never share a key. Inline code that the
ledger's storage would rewrite (CRLF, non-NFC text) is refused at mint: use `--file`.
-- **Near must also agree on behaviour.** A reworded match drops to `adapt`, with a note
+- **Only exact establishes the same task.** Every hit carries `semanticEquivalence`
+ (`"identical"` for exact, `"unverified"` for near and adapt) and `requiresReview` (`true`
+ for every non-exact hit). A near hit is a similar task, not a proven equivalent: review it
+ against yours before reusing it.
+- **Near must also agree on behaviour.** A similar spec drops to `adapt`, with a note
naming the difference, when anything behaviour-bearing differs, in content or in order:
operators with their operands, numbers, literals (typographic quotes included),
- identifiers, paths, negation, any whitespace other than one space, spelling (case,
- look-alike letters) or invisible characters. A key from an older version, or one the
- ledger stored normalized, never reaches near.
+ identifiers, paths, negation, what each polarity or direction word applies to ("allow
+ admins, deny guests" vs "deny admins, allow guests"; "from staging to production" vs the
+ reverse), any whitespace other than one space, spelling (case, look-alike letters) or
+ invisible characters. A key from an older version, or one the ledger stored normalized,
+ never reaches near.
- **Its dependencies are what its imports bind to.** A named import records its defining
module's declaration, followed through re-exports and export aliases. A default,
namespace or side-effect import, `require` or `import()` records the whole module's
diff --git a/mintlify/cli/quality.mdx b/mintlify/cli/quality.mdx
index deb5987..0aa0a90 100644
--- a/mintlify/cli/quality.mdx
+++ b/mintlify/cli/quality.mdx
@@ -37,12 +37,20 @@ script covers the workspaces in one run only when forge establishes, from its sh
structure, an unfiltered recursive run of every member's `test` script whose failure reaches
the exit status (`npm test --workspaces`, `pnpm -r test`, `turbo run test`, …). Filters,
masked runs and look-alike flags such as Node's preload `node -r` cover nothing, and each
-package then runs its own suite. Nothing that could narrow or replay the run is credited:
-options are allowlisted per tool, a script that changes shell state (`cd`, `exit`, `trap`) or
-sets `npm_config_*` is refused, `.npmrc`/lerna/nx/turbo config that filters the run is
-honoured, and turbo needs `--force` and nx/lerna `--skip-nx-cache` so a cached result is
-never replayed. `coverage basis` labels every verdict `measured`, `inferred` or `declared`,
-and a root command that was recognized but not credited is printed with the reason. Tune it
+package then runs its own suite. A tool counts only by its bare name or its own
+`node_modules/.bin` install: a program named by any other path (`./tools/npm`) is unknown,
+whatever its name. Nothing that could narrow, replace or replay the run is credited:
+- options are allowlisted per tool, and so are the variables the command sets (`PATH`,
+ `LD_PRELOAD` or `npm_config_*` are refused);
+- a script that changes shell state (`cd`, `exit`, `trap`) is refused;
+- a shim of the tool (a package manager in `node_modules/.bin`, another package's binary, a
+ same-named Windows executable in the root, a `yarnPath` outside `.yarn/releases/`) is
+ refused;
+- `.npmrc`/lerna/nx/turbo config that filters the run is honoured;
+- turbo needs `--force` and nx/lerna `--skip-nx-cache`, so a cached result is never replayed.
+
+`coverage basis` labels every verdict `measured`, `inferred` or `declared`, and a root
+command that was not credited is printed with the reason. Tune it
under `verify` in `.forge/forge.config.json`:
```json
diff --git a/mintlify/concepts/proof-carrying-memory.mdx b/mintlify/concepts/proof-carrying-memory.mdx
index 8ee2aaa..3ac7b5b 100644
--- a/mintlify/concepts/proof-carrying-memory.mdx
+++ b/mintlify/concepts/proof-carrying-memory.mdx
@@ -106,8 +106,12 @@ byte-identical to what was verified, and its dependencies must still resolve wit
declarations. Otherwise it falls through to generation and mints a fresh claim on the way
back. An exact hit needs the same text, byte for byte: no whitespace or Unicode
normalization, because the spaces in `"a b"` or a Python block's indentation are data. A
-near hit must also agree on operators, numbers, literals, identifiers, paths, negation and
-code layout, so `age >= 18` is never served for `age <= 18`.
+near hit must also agree on operators, numbers, literals, identifiers, paths, negation, what
+each polarity or direction word applies to, and code layout, so `age >= 18` is never served
+for `age <= 18`, nor "allow admins, deny guests" for "deny admins, allow guests". Only an
+exact hit establishes the same task (`semanticEquivalence: "identical"`). A near or adapt
+hit is a similar candidate whose equivalence is unverified, and it carries
+`requiresReview: true`.
One event is one vote. Abbreviations of one commit count once. A reworded lesson inherits
trust only when the rewrite is equivalent. Similar-but-opposite rules are reported as
diff --git a/mintlify/concepts/verification-gates.mdx b/mintlify/concepts/verification-gates.mdx
index d3ef59f..af847da 100644
--- a/mintlify/concepts/verification-gates.mdx
+++ b/mintlify/concepts/verification-gates.mdx
@@ -64,12 +64,20 @@ script covers the workspaces in one run only when forge establishes, from its sh
structure, an unfiltered recursive run of every member's `test` script whose failure reaches
the exit status (`npm test --workspaces`, `pnpm -r test`, `turbo run test`, …). Filters,
masked runs and look-alike flags such as Node's preload `node -r` cover nothing, and each
-package then runs its own suite. Nothing that could narrow or replay the run is credited:
-options are allowlisted per tool, a script that changes shell state (`cd`, `exit`, `trap`) or
-sets `npm_config_*` is refused, `.npmrc`/lerna/nx/turbo config that filters the run is
-honoured, and turbo needs `--force` and nx/lerna `--skip-nx-cache` so a cached result is
-never replayed. `coverage basis` labels every verdict `measured`, `inferred` or `declared`,
-and a root command that was recognized but not credited is printed with the reason. Tune it
+package then runs its own suite. A tool counts only by its bare name or its own
+`node_modules/.bin` install: a program named by any other path (`./tools/npm`) is unknown,
+whatever its name. Nothing that could narrow, replace or replay the run is credited:
+- options are allowlisted per tool, and so are the variables the command sets (`PATH`,
+ `LD_PRELOAD` or `npm_config_*` are refused);
+- a script that changes shell state (`cd`, `exit`, `trap`) is refused;
+- a shim of the tool (a package manager in `node_modules/.bin`, another package's binary, a
+ same-named Windows executable in the root, a `yarnPath` outside `.yarn/releases/`) is
+ refused;
+- `.npmrc`/lerna/nx/turbo config that filters the run is honoured;
+- turbo needs `--force` and nx/lerna `--skip-nx-cache`, so a cached result is never replayed.
+
+`coverage basis` labels every verdict `measured`, `inferred` or `declared`, and a root
+command that was not credited is printed with the reason. Tune it
under `verify` in `.forge/forge.config.json`:
```json
diff --git a/scripts/claims-status.mjs b/scripts/claims-status.mjs
index ddc597c..c3f68a4 100644
--- a/scripts/claims-status.mjs
+++ b/scripts/claims-status.mjs
@@ -190,17 +190,33 @@ export function validateRegistry(registry, { root = null } = {}) {
return errors;
}
+/** Compare two `x.y.z` versions numerically (a pre-release suffix is ignored). */
+export function compareVersions(a, b) {
+ const parts = (v) =>
+ String(v)
+ .split(/[.+-]/)
+ .slice(0, 3)
+ .map((n) => Number.parseInt(n, 10) || 0);
+ const [x, y] = [parts(a), parts(b)];
+ for (let i = 0; i < 3; i++) if (x[i] !== y[i]) return x[i] - y[i];
+ return 0;
+}
+
/**
* Cross-check each claim's `assessed_release` against the checkout's release tags: a claim
* marked "unreleased" whose source commit has since shipped is stale (record the release), and
* a claim naming a release must name one whose tag exists and contains its source commit.
* Returns null when the check cannot run — no git, or no `v*` tags (a shallow CI clone) — and
- * skips a claim whose commit is not in this clone.
+ * skips a claim whose commit is not in this clone. A checkout can also carry only SOME tags (a
+ * branch fetched without `--tags`): a release newer than every tag present but not newer than
+ * the code's own package.json version cannot be checked there — it is listed in `unchecked`
+ * (fetch the tags to check it), not reported as a problem.
* @param {string} root
* @param {any} registry a valid registry
+ * @param {{unchecked?: string[]}} [opts] receives the claims that could not be checked
* @returns {string[]|null}
*/
-export function releaseProblems(root, registry) {
+export function releaseProblems(root, registry, { unchecked = [] } = {}) {
const g = (args) =>
execFileSync("git", args, {
cwd: root,
@@ -222,6 +238,14 @@ export function releaseProblems(root, registry) {
return null;
}
if (!tags.size) return null;
+ const newest = [...tags]
+ .map((t) => t.slice(1))
+ .sort(compareVersions)
+ .at(-1);
+ let version = null;
+ try {
+ version = JSON.parse(readFileSync(path.join(root, "package.json"), "utf8")).version ?? null;
+ } catch {}
const firstRelease = new Map();
const shippedIn = (commit) => {
if (!firstRelease.has(commit)) {
@@ -247,9 +271,15 @@ export function releaseProblems(root, registry) {
`${c.id}: assessed_release is "unreleased", but its source commit ${at} shipped in ${first} — record the release`,
);
} else if (!tags.has(`v${c.assessed_release}`)) {
- problems.push(
- `${c.id}: assessed_release ${c.assessed_release} has no v${c.assessed_release} tag`,
- );
+ const untagged =
+ typeof version === "string" &&
+ compareVersions(c.assessed_release, newest) > 0 &&
+ compareVersions(c.assessed_release, version) <= 0;
+ if (untagged) unchecked.push(`${c.id} (${c.assessed_release})`);
+ else
+ problems.push(
+ `${c.id}: assessed_release ${c.assessed_release} has no v${c.assessed_release} tag`,
+ );
} else if (!ok(["merge-base", "--is-ancestor", c.source_commit, `v${c.assessed_release}`])) {
problems.push(
`${c.id}: release ${c.assessed_release} does not contain its source commit ${at}`,
@@ -421,12 +451,18 @@ export function run(argv, io = {}) {
for (const p of problems) error(`registry: ${p}`);
return 1;
}
- const releases = releaseProblems(root, registry);
+ /** @type {string[]} */
+ const unchecked = [];
+ const releases = releaseProblems(root, registry, { unchecked });
if (releases === null) log("release check skipped: no v* tags in this checkout");
else if (releases.length) {
for (const p of releases) error(`registry: ${p}`);
return 1;
}
+ if (unchecked.length)
+ log(
+ `release check incomplete: ${unchecked.length} claim(s) name a release newer than this checkout's tags — \`git fetch --tags\` to check ${unchecked.join(", ")}`,
+ );
let failed = false;
const readmeFile = path.join(root, README_PATH);
diff --git a/src/cli/memory.js b/src/cli/memory.js
index 1241e4e..cd76173 100644
--- a/src/cli/memory.js
+++ b/src/cli/memory.js
@@ -502,6 +502,9 @@ HANDLERS.reuse = async (argv) => {
sim: r.sim,
revalidation: r.revalidation?.status,
requiresRevalidation: r.requiresRevalidation === true,
+ ...(r.tier === "miss"
+ ? {}
+ : { semanticEquivalence: r.semanticEquivalence, requiresReview: r.requiresReview }),
reasons: r.reasons,
},
null,
@@ -520,9 +523,13 @@ HANDLERS.reuse = async (argv) => {
` claim ${a.id.slice(0, 12)} — \`forge ledger blame ${a.id.slice(0, 8)}\` for its proof`,
);
if (r.tier === "near")
- console.log(" near tier: a reworded match — review the diff before reusing it as-is");
+ console.log(
+ " near tier: a similar task, equivalence unverified — review it against yours before reusing",
+ );
if (r.tier === "adapt")
- console.log(" adapt tier: inject as a verified starting point, generate only the delta");
+ console.log(
+ " adapt tier: a verified starting point for a similar task — generate only the delta, and review it",
+ );
if (r.requiresRevalidation)
console.log(
` NOT revalidated: ${(r.revalidation?.unknown ?? []).join(", ")} — check before use`,
diff --git a/src/reuse.js b/src/reuse.js
index 68731e0..502ef2d 100644
--- a/src/reuse.js
+++ b/src/reuse.js
@@ -711,8 +711,11 @@ export function revalidate(artifact, atlas, { root = null } = {}) {
* null for that candidate (missing vector), MinHash Jaccard with NEAR_J/ADAPT_J is the
* per-candidate fallback — a partially-embedded ledger never loses lexical recall.
* Similarity alone never serves code as-is (review F04): a near candidate whose operators,
- * numbers, literals, identifiers, paths or polarity words differ from the query is held at
- * the adapt tier (a starting point to review), with the conflict named in `reasons`.
+ * numbers, literals, identifiers, paths, polarity words or their bindings differ from the
+ * query is held at the adapt tier (a starting point to review), with the conflict named in
+ * `reasons`. And similarity never establishes equivalence (review Q01): every hit carries
+ * `semanticEquivalence` — "identical" for exact (the byte-identical spec), "unverified" for
+ * near and adapt — and `requiresReview`, true for every non-exact hit.
* Every hit is revalidated at the serving boundary (review F05, see revalidate): an
* `invalid` artifact is never served; an `unknown` one is returned with
* `requiresRevalidation: true` — never presented as checked.
@@ -722,6 +725,7 @@ export function revalidate(artifact, atlas, { root = null } = {}) {
* sim?:((query:any, claim:any)=>number|null)|null}} opts
* @returns {{tier:"exact"|"near"|"adapt"|"miss", artifact?:any, jaccard?:number,
* similarity?:number, simBackend?:string, revalidation?:object,
+ * semanticEquivalence?:"identical"|"unverified", requiresReview?:boolean,
* requiresRevalidation?:boolean, reasons:string[], sim?:string,
* invalidated?:{id:string, missing:string[], changed:string[], problems:string[]}[]}}
* `sim` is stamped by reuseQuery/reusePeek (the backend label the CLI prints);
@@ -758,10 +762,19 @@ export function lookup(
reasons.push(`${why} ${c.id.slice(0, 8)} failed revalidation: ${rv.problems.join("; ")}`);
return null;
};
+ // Review Q01: only the exact tier ESTABLISHES that the task is the one the artifact was
+ // verified for (the same spec, byte for byte). A near or adapt hit is a similar candidate —
+ // no check here can establish that two texts mean the same — so it is always marked as one
+ // that requires review, with its equivalence unverified. `revalidation` is a separate
+ // question: whether the artifact and its dependencies still hold, not whether it fits.
const served = (tier, c, rv, extra = {}) => ({
tier,
artifact: c,
...extra,
+ semanticEquivalence: /** @type {"identical"|"unverified"} */ (
+ tier === "exact" ? "identical" : "unverified"
+ ),
+ requiresReview: tier !== "exact",
revalidation: rv,
...(rv.status === "valid" ? {} : { requiresRevalidation: true }),
reasons,
@@ -796,7 +809,7 @@ export function lookup(
const qBands = new Set(bandKeys(qs));
pool = artifacts.filter((c) => bandKeys(shapeOf(c)).some((k) => qBands.has(k)));
}
- // near compares IDENTITY (same names, reworded prose); adapt compares SHAPE too, so
+ // near compares IDENTITY (same names, similar prose); adapt compares SHAPE too, so
// `add pagination to listOrders` can still be offered the listUsers artifact as a
// starting point — the tier that says "generate only the delta" — but never as-is.
const measure = (c) => {
diff --git a/src/semantic_guard.js b/src/semantic_guard.js
index b850f00..a69266f 100644
--- a/src/semantic_guard.js
+++ b/src/semantic_guard.js
@@ -430,13 +430,159 @@ function spellingsOf(masked) {
return map;
}
+// ---------------------------------------------------------------------------------------
+// Relational binding (review Q01). "Allow admins and deny guests…" and "Deny admins and allow
+// guests…" hold the same tokens — the same polarity words, the same names — and score as
+// near-identical; what differs is WHICH word each polarity or direction word applies to. Each
+// relation word (a negator, a pole of an antonym pair, a direction preposition) is bound to
+// the next content word of its clause, a quoted literal included. Two texts conflict on
+// `binding` when
+// 1. a word both texts bind is bound to OPPOSITE relations (allow→admins vs deny→admins,
+// from→staging vs to→staging; a negator bound to the same word flips its poles), or
+// 2. they use exactly the same words in a different order — a rearrangement, which can swap
+// who is allowed, or source and destination, without changing a single token.
+// A token heuristic, not a parser: it holds known reversals back from the near tier. It
+// cannot establish that two texts mean the same — nothing in this module can — which is why
+// every non-exact reuse hit is a candidate that requires review (reuse.js), never an
+// equivalent.
+// ---------------------------------------------------------------------------------------
+
+/** word → every [relation class, pole] it expresses (antonym pairs, then direction). */
+const RELATION_POLES = new Map();
+/** @param {Iterable} words @param {string} cls @param {number} pole */
+const addPoles = (words, cls, pole) => {
+ for (const w of words) RELATION_POLES.set(w, [...(RELATION_POLES.get(w) ?? []), [cls, pole]]);
+};
+ANTONYM_PAIRS.forEach(([a, b], i) => {
+ addPoles(a, `p${i}`, 0);
+ addPoles(b, `p${i}`, 1);
+});
+addPoles(["from"], "direction", 0);
+addPoles(["to", "into", "onto", "toward", "towards"], "direction", 1);
+// Function words passed over when looking for the word a relation applies to.
+const BIND_SKIP = new Set(
+ (
+ "a an the all any every each some only just also both either its it their them they our " +
+ "your his her my this that these those same other such be is are was were been being am " +
+ "do does did have has had will would shall should can could might of for with by at in as " +
+ "per via about up down out"
+ ).split(" "),
+);
+// Words that end the phrase a pending relation can apply to.
+const CLAUSE_WORDS = new Set(
+ "and or but then while whereas unless except if when so yet".split(" "),
+);
+const isNegator = (w) => NEGATORS.has(w) || NEGATED_CONTRACTION.test(w);
+const CLAUSE_END = new Set([...",;:.!?"]);
+const CLOSING = new Set([..."\"'”’)]"]);
+
+/**
+ * The words of a masked text (lower-cased, a possessive `'s` dropped; a literal is its own
+ * text) and each relation
+ * word's binding `rel→object`, both in document order.
+ * @param {string} masked
+ * @param {string[]} literals
+ * @returns {{words: string[], bindings: {rel: string, obj: string}[]}}
+ */
+function relationsOf(masked, literals) {
+ /** @type {string[]} */
+ const words = [];
+ /** @type {{rel: string, obj: string}[]} */
+ const bindings = [];
+ /** @type {string[]} */
+ let pending = [];
+ let lit = 0;
+ for (const raw of masked.split(/\s+/u)) {
+ if (!raw) continue;
+ const marks = [...raw].filter((ch) => ch === LIT).length;
+ const quoted = marks ? literals.slice(lit, lit + marks).join(" ") : "";
+ lit += marks;
+ // a possessive names the same party: "the author's change" binds `author`
+ const word =
+ quoted ||
+ trimEdges(raw.replaceAll(FENCE, ""))
+ .toLowerCase()
+ .replace(/['’]s$/u, "");
+ if (!word || !/[\p{L}\p{N}]/u.test(word)) {
+ pending = []; // a dash, a bare symbol: the phrase ends
+ continue;
+ }
+ words.push(word);
+ if (!quoted && (isNegator(word) || RELATION_POLES.has(word))) pending.push(word);
+ else if (!quoted && CLAUSE_WORDS.has(word)) pending = [];
+ else if (quoted || !BIND_SKIP.has(word)) {
+ for (const rel of pending) bindings.push({ rel, obj: word });
+ pending = [];
+ }
+ let end = raw.length;
+ while (end > 0 && CLOSING.has(raw[end - 1])) end -= 1;
+ if (end > 0 && CLAUSE_END.has(raw[end - 1])) pending = []; // `admins,` ends the phrase
+ }
+ return { words, bindings };
+}
+
+/** Each bound word's relation classes, with the pole a negator on the same word flips. */
+function relationSignatures(bindings) {
+ /** @type {Map} */
+ const byObj = new Map();
+ for (const { rel, obj } of bindings) {
+ const rels = byObj.get(obj);
+ if (rels) rels.push(rel);
+ else byObj.set(obj, [rel]);
+ }
+ /** @type {Map>} */
+ const out = new Map();
+ for (const [obj, rels] of byObj) {
+ const flip = rels.filter(isNegator).length % 2;
+ const sig = new Set();
+ for (const rel of rels)
+ for (const [cls, pole] of RELATION_POLES.get(rel) ?? []) sig.add(`${cls}:${pole ^ flip}`);
+ out.set(obj, sig);
+ }
+ return out;
+}
+
+/**
+ * The `binding` conflict between two texts' relations (see above), or null.
+ * @param {{words: string[], bindings: {rel: string, obj: string}[]}} A
+ * @param {{words: string[], bindings: {rel: string, obj: string}[]}} B
+ * @returns {{kind: string, a: string[], b: string[], order?: boolean}|null}
+ */
+function bindingConflict(A, B) {
+ const shown = (bs) => bs.map((x) => `${x.rel}→${x.obj}`);
+ // 1. a word both bind, bound to opposite poles of one relation
+ const [sa, sb] = [relationSignatures(A.bindings), relationSignatures(B.bindings)];
+ const opposed = new Set();
+ const opposite = (x, y) =>
+ [...x].some((k) => !y.has(k) && y.has(`${k.slice(0, -1)}${1 - Number(k.slice(-1))}`));
+ for (const [obj, x] of sa) {
+ const y = sb.get(obj);
+ if (y && (opposite(x, y) || opposite(y, x))) opposed.add(obj);
+ }
+ if (opposed.size)
+ return {
+ kind: "binding",
+ a: shown(A.bindings.filter((x) => opposed.has(x.obj))),
+ b: shown(B.bindings.filter((x) => opposed.has(x.obj))),
+ };
+ // 2. the same words, rearranged
+ if (sameList(A.words, B.words) || !sameList(sorted(A.words), sorted(B.words))) return null;
+ let i = 0;
+ while (A.words[i] === B.words[i]) i += 1;
+ let j = 0;
+ while (A.words[A.words.length - 1 - j] === B.words[B.words.length - 1 - j]) j += 1;
+ const span = (ws) => ws.slice(i, ws.length - j);
+ const clip = (ws) => (ws.length > 12 ? [...ws.slice(0, 12), "…"] : ws);
+ return { kind: "binding", a: clip(span(A.words)), b: clip(span(B.words)), order: true };
+}
+
/**
* The behaviour-carrying features of a text, each in DOCUMENT ORDER (review N01 round 2:
* sorted lists made `a - b` and `b - a`, or two swapped indented lines, the same text).
* @param {string} text
* @returns {{operators: string[], numbers: string[], literals: string[],
- * identifiers: string[], paths: string[], polarity: string[], layout: string[],
- * symbols: string[], format: string[]}}
+ * identifiers: string[], paths: string[], polarity: string[], bindings: string[],
+ * layout: string[], symbols: string[], format: string[]}}
*/
export function criticalFeatures(text) {
return analyze(text).features;
@@ -478,6 +624,7 @@ function analyze(text) {
`${cf.map((ch) => `U+${(ch.codePointAt(0) ?? 0).toString(16).toUpperCase().padStart(4, "0")}`).join(" ")} in ${JSON.stringify(tok)}`,
);
}
+ const relations = relationsOf(masked, literals);
const features = {
operators,
numbers,
@@ -485,11 +632,12 @@ function analyze(text) {
identifiers,
paths,
polarity: sorted(polarity),
+ bindings: relations.bindings.map((x) => `${x.rel}→${x.obj}`),
layout: [...fences.map((f) => `fence ${JSON.stringify(f)}`), ...whitespaceItems(masked)],
symbols: symbolRuns(masked),
format,
};
- return { features, masked };
+ return { features, masked, relations };
}
const sameList = (a, b) => a.length === b.length && a.every((x, i) => x === b[i]);
@@ -517,6 +665,7 @@ export function polarityFlip(a, b) {
* a rewrite that names a new identifier or path is more specific, not the opposite). */
export const ALL_KINDS = /** @type {const} */ ([
"polarity",
+ "binding",
"operators",
"numbers",
"literals",
@@ -529,6 +678,7 @@ export const ALL_KINDS = /** @type {const} */ ([
]);
export const FLIP_KINDS = /** @type {const} */ ([
"polarity",
+ "binding",
"operators",
"numbers",
"literals",
@@ -551,6 +701,11 @@ export function semanticConflicts(a, b, { kinds = ALL_KINDS } = {}) {
const [fa, fb] = [A.features, B.features];
const out = [];
for (const kind of kinds) {
+ if (kind === "binding") {
+ const c = bindingConflict(A.relations, B.relations);
+ if (c) out.push(c);
+ continue;
+ }
if (kind === "spelling") {
const sa = spellingsOf(A.masked);
const sb = spellingsOf(B.masked);
diff --git a/src/stack.js b/src/stack.js
index fd5fb40..e8a5098 100644
--- a/src/stack.js
+++ b/src/stack.js
@@ -4,8 +4,8 @@
// Everything is data: SIGNATURES maps a dependency/marker to a label, so widening coverage
// is adding a row, never editing logic. Fail-safe — an unreadable or absent manifest is
// skipped, never thrown.
-import { existsSync, readdirSync, readFileSync } from "node:fs";
-import { join } from "node:path";
+import { existsSync, lstatSync, readdirSync, readFileSync, realpathSync } from "node:fs";
+import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
import { stripTrailingSlashes } from "./util.js";
const read = (root, rel) => {
@@ -622,7 +622,42 @@ const CACHE_BYPASS_ENV = new Map([
["TURBO_FORCE", "turbo"],
["NX_SKIP_NX_CACHE", "nx"],
]);
+// Variables a recognized command line may set — an ALLOWLIST, like the options (review Q02).
+// Any other can change which program a name runs (PATH), what it loads (LD_PRELOAD,
+// NODE_OPTIONS=--require), where it reads its configuration (HOME), or which members it runs.
+const INERT_ENV = new Set([
+ "CI",
+ "NODE_ENV",
+ "FORCE_COLOR",
+ "NO_COLOR",
+ "TZ",
+ "LANG",
+ "LC_ALL",
+ "DEBUG",
+ "DO_NOT_TRACK",
+ "NX_DAEMON",
+ "NX_NO_CLOUD",
+ "TURBO_TELEMETRY_DISABLED",
+]);
+// NODE_OPTIONS flags that neither load code nor soften a failure.
+const INERT_NODE_OPTION =
+ /^--(?:(?:max[-_]old[-_]space[-_]size|max[-_]semi[-_]space[-_]size|stack[-_]trace[-_]limit)=\d+|experimental-vm-modules|no-warnings|no-deprecation|trace-warnings|trace-deprecation|trace-uncaught|enable-source-maps|unhandled-rejections=(?:strict|throw))$/;
const TRUTHY = new Set(["1", "true"]);
+
+/** Whether a NODE_OPTIONS value holds only inert flags. */
+const inertNodeOptions = (value) =>
+ String(value)
+ .split(/\s+/)
+ .every((o) => !o || INERT_NODE_OPTION.test(o));
+
+/** Why a variable set on the command line keeps a run from being established, or null. */
+function envProblem(name, value) {
+ if (INERT_ENV.has(name) || CACHE_BYPASS_ENV.has(name)) return null;
+ if (name === "NODE_OPTIONS" && inertNodeOptions(value)) return null;
+ if (/^path(?:ext)?$/i.test(name)) return `sets ${name}, which changes the program a name runs`;
+ if (TOOL_ENV.test(name)) return `sets ${name}, which can narrow or redirect the run`;
+ return `sets ${name}, which is not a variable known to leave the run unchanged`;
+}
// Shell builtins that change what LATER commands do — the directory (`cd packages/good && npm
// test --workspaces` runs one workspace), the exit status (`trap 'exit 0' EXIT`, `… && exit
// 0; …`), the environment, the meaning of a name. A script holding any is not a recognized
@@ -635,6 +670,8 @@ const STATEFUL_BUILTINS = new Set(
);
// Wrappers whose job is to find and run another binary.
const TOOL_BINS = new Set(["turbo", "lerna", "nx"]);
+// npx/bunx/npm exec options that change nothing about which binary runs. `-p`/`--package`
+// is not one: it picks the package that provides the binary (review Q02).
const NPX_OK = new Set([
"--yes",
"-y",
@@ -644,21 +681,65 @@ const NPX_OK = new Set([
"-q",
"--prefer-offline",
]);
-/** `./node_modules/.bin/turbo.cmd` → `turbo`. */
-const binName = (w) =>
+// Programs a project installs as a dependency, and the package whose binary each must be. A
+// bare name finds them through the script's PATH (node_modules/.bin first); the same install
+// may be spelled out as `node_modules/.bin/`. Package managers and system programs are
+// never project binaries (runProblem refuses a copy in node_modules/.bin).
+const INSTALLED_BINS = new Map([
+ ["turbo", "turbo"],
+ ["nx", "nx"],
+ ["lerna", "lerna"],
+ ["cross-env", "cross-env"],
+ ["bun", "bun"],
+ ["bunx", "bun"],
+]);
+
+/**
+ * The program a command word names, as far as recognition may rely on it (review Q02). A bare
+ * name is the program of that name the script's PATH resolves — runProblem checks that nothing
+ * in the project shadows it. `node_modules/.bin/` (optionally `./`-prefixed, or its
+ * Windows `.cmd`/`.ps1` shim) is that same installed tool spelled out. Any other path —
+ * `./tools/npm`, `./scripts/pnpm.js`, `/usr/bin/env` — is a program whose behaviour is
+ * unknown, whatever its basename: null.
+ * @param {string} word
+ * @returns {string|null}
+ */
+function commandIdentity(word) {
+ const w = String(word);
+ if (!/[\\/]/.test(w)) return w;
+ const m = /^(?:\.[\\/])?node_modules[\\/]\.bin[\\/]([^\\/]+?)(?:\.cmd|\.ps1)?$/i.exec(w);
+ return m && INSTALLED_BINS.has(m[1]) ? m[1] : null;
+}
+
+/** The name a word's basename suggests (`./tools/npm.cmd` → `npm`) — used only to explain a
+ * refusal, never to recognize a program. */
+const suggestedName = (w) =>
String(w)
.split(/[\\/]/)
.pop()
- .replace(/\.(?:cmd|exe|ps1|js|cjs|mjs)$/i, "");
+ ?.replace(/\.(?:cmd|exe|ps1|bat|js|cjs|mjs)$/i, "") ?? "";
-/** Strip env assignments and runner wrappers (env, cross-env, npx, bunx, npm exec, pnpm/yarn
- * exec|dlx, `yarn turbo`) down to the executable that does the work, with the variables the
- * command line sets. `null` when a wrapper is used in a way that could change what runs
- * (`npx -c`, `npm exec --workspaces`…) or a tool-namespaced variable is set. */
-function unwrapCommand(words) {
+/**
+ * Strip env assignments and runner wrappers (env, cross-env, npx, bunx, npm exec, pnpm/yarn
+ * exec|dlx, `yarn turbo`) down to the program that does the work: its identity (`bin`), its
+ * arguments, the variables the command line sets, and `bins` — every program the line relies
+ * on, wrappers first, for runProblem's shadow checks. `null` when a wrapper is used in a way
+ * that could change what runs (`npx -c`, `npx -p pkg`, `env -C dir`, `npm exec
+ * --workspaces`…), a word names an unknown program (commandIdentity), or a variable outside
+ * the allowlist is set. `explain` relaxes the last two and records the first such reason as
+ * `note` instead — for a refusal message only.
+ * @param {string[]} words
+ * @param {boolean} [explain]
+ * @returns {{bin: string, args: string[], words: string[], env: Map, bins: string[], note: string|null}|null}
+ */
+function unwrapCommand(words, explain = false) {
const w = [...words];
/** @type {Map} */
const env = new Map();
+ /** @type {string[]} */
+ const bins = [];
+ /** @type {string|null} */
+ let note = null;
const assign = () => {
while (w.length && ENV_ASSIGN.test(w[0])) {
const word = /** @type {string} */ (w.shift());
@@ -666,16 +747,29 @@ function unwrapCommand(words) {
env.set(word.slice(0, eq), word.slice(eq + 1));
}
};
+ /** @param {string} word */
+ const identify = (word) => {
+ const id = commandIdentity(word);
+ if (id !== null || !explain) return id;
+ const name = suggestedName(word);
+ note ??= `\`${word}\` is a program named by its path, not ${name} itself: only a bare name or a node_modules/.bin install is recognized`;
+ return name;
+ };
for (let hop = 0; hop < 6; hop++) {
assign();
if (!w.length || w[0] === "!") return null; // `! cmd` inverts the status
- const bin = binName(w[0]);
+ const bin = identify(w[0]);
+ if (bin === null) return null;
+ bins.push(bin);
const next = w[1];
if (bin === "env" || bin === "cross-env") {
w.shift();
+ // `env -u NAME` only removes a variable; `-i`, `-C`/`--chdir`, `-S`… change the
+ // environment or the directory the command runs in. cross-env takes no options.
while (w.length && w[0].startsWith("-")) {
- if (w[0] === "-u" || w[0] === "--unset") w.shift();
- w.shift();
+ if (bin === "env" && (w[0] === "-u" || w[0] === "--unset") && w.length > 1) w.splice(0, 2);
+ else if (bin === "env" && w[0].startsWith("--unset=")) w.shift();
+ else return null;
}
continue;
}
@@ -684,9 +778,8 @@ function unwrapCommand(words) {
if (bin === "bun") w.shift();
while (w.length && w[0].startsWith("-")) {
const o = /** @type {string} */ (w.shift());
- if (o === "-p" || o === "--package") w.shift();
- else if (o === "--") break;
- else if (!NPX_OK.has(o) && !o.startsWith("--package=")) return null;
+ if (o === "--") break;
+ if (!NPX_OK.has(o)) return null;
}
continue;
}
@@ -695,7 +788,7 @@ function unwrapCommand(words) {
while (w.length && w[0].startsWith("-")) {
const o = /** @type {string} */ (w.shift());
if (o === "--") break;
- if (!NPX_OK.has(o) && !o.startsWith("--package=")) return null;
+ if (!NPX_OK.has(o)) return null;
}
continue;
}
@@ -705,13 +798,22 @@ function unwrapCommand(words) {
if (w[0]?.startsWith("-")) return null;
continue;
}
- if ((bin === "pnpm" || bin === "yarn") && next && TOOL_BINS.has(binName(next))) {
+ // `yarn turbo run test`: the package manager runs the installed tool, named bare.
+ if (
+ (bin === "pnpm" || bin === "yarn") &&
+ next !== undefined &&
+ TOOL_BINS.has(explain ? suggestedName(next) : next)
+ ) {
w.shift();
continue;
}
- for (const name of env.keys())
- if (TOOL_ENV.test(name) && !CACHE_BYPASS_ENV.has(name)) return null;
- return { bin, args: w.slice(1), words: w, env };
+ for (const [name, value] of env) {
+ const why = envProblem(name, value);
+ if (!why) continue;
+ if (!explain) return null;
+ note ??= `the command line ${why}`;
+ }
+ return { bin, args: w.slice(1), words: w, env, bins, note };
}
return null;
}
@@ -933,37 +1035,71 @@ const MAX_SCRIPT_HOPS = 4;
const firstWord = (c) => c.words.find((w) => !ENV_ASSIGN.test(w)) ?? "";
/**
- * The recursive workspace test run a root `test` script performs, if one is ESTABLISHED:
- * `{tool, command, bypass}` for the recognized invocation (`command` is its words, re-joined;
- * `bypass` — the command itself disables the tool's result cache: `turbo --force`,
- * `--skip-nx-cache`, `TURBO_FORCE=1`), else null. Pure — `scripts` is the root package.json's
- * `scripts`, used to follow `npm run