From cf890405dc9f45b99cc87304efc857ef9921addd Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 05:11:15 +0000 Subject: [PATCH 1/3] fix: credit a workspace run only to the tool itself; mark non-exact reuse as needing review The 2026-09-27 recheck of v1.7.3 confirmed N01-N08 closed and found two adjacent cases. Both are fixed conservatively: what cannot be established runs the members on their own, or is served as a candidate to review. Q02 (P1), workspace coverage: `./tools/npm test --workspaces` was credited as npm's recursive run because recognition read only the file's basename, so a stub that exited 0 hid a failing workspace behind a PASS. - a tool is recognized by its bare name or its node_modules/.bin install; any other path is an unknown program, and the refusal is reported - variables on the command line are allowlisted (PATH, LD_PRELOAD, HOME and a NODE_OPTIONS preload are refused); npx -p and env -i/-C refused - shadows are refused: a package manager or system program in any node_modules/.bin on the script PATH, another package's binary (or a link out of the package) under a tool's name, a same-named Windows executable in the root, a yarnPath outside .yarn/releases, and a packageManager fetched from a URL - node-options must be inert, a script-shell inside the project is not a shell, and nx plugins run the members on their own - a verify.workspaces "root" declaration stays labelled declared Q01 (P2), near reuse: "Deny admins and allow guests..." near-hit an artifact verified for "Allow admins and deny guests..." (MinHash 0.83) and was called a "reworded match". - every hit carries semanticEquivalence ("identical" only for exact, "unverified" otherwise) and requiresReview (true unless exact), in `forge reuse query --json` and the gate's reuse summary - the semantic guard's new `binding` kind holds opposite bindings (allow->admins vs deny->admins, from/to swaps) and same-word rearrangements at adapt; consolidation lists such pairs as conflicts - the CLI, the gate and the docs no longer call a near hit reworded The claims check no longer fails in a checkout that lacks newer release tags: such a claim is reported unchecked (fetch the tags), not failed. Docs (GUIDE, Mintlify, the reuse plan), CHANGELOG and the rendered changelog page are updated in the same change. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- CHANGELOG.md | 51 +++ docs/GUIDE.md | 44 ++- docs/plans/substrate-v2/03-reuse-cache.md | 7 +- mintlify/changelog/overview.mdx | 12 + mintlify/cli/memory.mdx | 14 +- mintlify/cli/quality.mdx | 20 +- mintlify/concepts/proof-carrying-memory.mdx | 8 +- mintlify/concepts/verification-gates.mdx | 20 +- scripts/claims-status.mjs | 48 ++- src/cli/memory.js | 11 +- src/reuse.js | 19 +- src/semantic_guard.js | 161 ++++++++- src/stack.js | 376 +++++++++++++++++--- src/substrate.js | 13 +- test/claims_status.test.js | 40 +++ test/reuse.test.js | 46 +++ test/semantic_guard.test.js | 75 +++- test/stack.test.js | 247 ++++++++++++- test/trust_properties.test.js | 26 +- test/verify.test.js | 43 +++ 20 files changed, 1178 insertions(+), 103 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b5a2fb0..345cf8f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,57 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +Fixes for the two findings of the 2026-09-27 recheck of v1.7.3 (Q01, Q02). The recheck +confirmed that all eight N01–N08 reproduction cases are closed; its scripts +(`reproduce_remaining.mjs`, `original-reproduce.mjs` and `edge-probes.mjs`, each with +`--assert-fixed`) all pass. + +### Fixed + +- **A local program named like a package manager no longer covers the workspaces (Q02).** A + root `test` script of `./tools/npm test --workspaces` was credited as npm's recursive run + because the recognizer read only the file's basename, so a stub that exited 0 hid a failing + workspace behind a PASS. A tool is now recognized only by its bare name (the program the + script's PATH finds) or as its own install under `node_modules/.bin`. Any other path, such as + `./tools/npm`, `./scripts/pnpm.js` or `/usr/bin/env`, is an unknown program whatever its + name, so the members run on their own. `forge verify` reports the root run as "not credited" + and names the program. + - A command line may set only variables known to change neither the program nor the run + (`CI`, `NODE_ENV`, `FORCE_COLOR`, `NODE_OPTIONS` with memory or warning flags…). `PATH`, + `LD_PRELOAD`, `HOME`, a `NODE_OPTIONS` preload and any other variable keep the run from + being credited. + - Wrapper options that pick the binary, the directory or the environment are refused: + `npx -p`/`--package`, `env -i`, `env -C`/`--chdir`, and options on `cross-env`. + - A bare name must not be shadowed. The first `node_modules/.bin` entry on the script's PATH + (the package's own, then each parent directory's) must be the tool's own package's binary, + and a symlink must resolve into that package. A package manager or system program there is + a shim. On Windows, a same-named executable in the package root (`npm.cmd`) runs first, so + it counts as a shadow too. + - yarn must be a release: a `yarnPath` (or yarn 1's `yarn-path`) outside `.yarn/releases/`, + or a `packageManager` that fetches the manager from a URL, is refused. + - `.npmrc` and environment `node-options` must be inert, a `script-shell` that is a program + inside the project is not a shell, and nx plugins, which can redefine each project's `test` + target, make nx and lerna runs run the members on their own. + - A `verify.workspaces: "root"` declaration still covers every member and is still labelled + `declared`. +- **A near reuse hit is no longer presented as the same task (Q01).** "Deny admins and allow + guests…" near-hit an artifact verified for "Allow admins and deny guests…" (MinHash 0.83), + and the CLI called it a "reworded match". + - Every hit now says what it establishes. `semanticEquivalence` is `"identical"` for an + exact hit (the byte-identical spec) and `"unverified"` for near and adapt, and + `requiresReview` is `true` for every non-exact hit. `forge reuse query --json` and the gate's + reuse summary carry both fields, and the CLI and gate describe a near hit as "a similar + task, equivalence unverified". + - The semantic guard adds a `binding` conflict kind. Each polarity, negation or direction + word binds to the next content word of its clause. Two texts conflict when a word both bind + is bound to opposite relations (allow→admins vs deny→admins, from→staging vs to→staging), or + when they use exactly the same words in a different order. Permission-subject swaps, + source/destination swaps and role swaps are therefore held at `adapt`. Consolidation and + compaction list such pairs as conflicts rather than proposals. +- **The claims check no longer fails in a checkout that lacks newer release tags.** A claim + naming a release newer than every local tag, and not newer than `package.json`'s version, is + reported as unchecked, with a hint to run `git fetch --tags`, instead of failing the check. + ## [1.7.3] - 2026-09-27 Fixes for the eight findings of the 2026-09-27 follow-up review (N01–N08) and its suggestions. diff --git a/docs/GUIDE.md b/docs/GUIDE.md index 0ef7537..0e38e84 100644 --- a/docs/GUIDE.md +++ b/docs/GUIDE.md @@ -705,11 +705,24 @@ conservative too: - Options are allowlisted per tool. An unknown one (`--help`, `--dry-run`, an abbreviation npm would expand) is not credited, and neither are words after `--`, which are forwarded into every member's script. -- A script that changes shell state (`cd`, `exit`, `trap`, `export`…) or sets a - package-tool variable (`npm_config_*`) is not credited. +- A tool is recognized by its bare name, the program the script's PATH finds, or as its own + install under `node_modules/.bin`. Any other path (`./tools/npm`, `./scripts/pnpm.js`, + `/usr/bin/env`) is an unknown program whatever its name, and is never credited. +- A script that changes shell state (`cd`, `exit`, `trap`, `export`…) is not credited, and + the command may set only variables known to change neither the program nor the run (`CI`, + `NODE_ENV`, `FORCE_COLOR`, `NODE_OPTIONS` with memory or warning flags, the cache-bypass + variables…). `PATH`, `LD_PRELOAD`, `HOME`, `npm_config_*` and anything else are refused, and + so are wrapper options that pick the binary or the directory (`npx -p`, `env -C`). +- A bare name must not be shadowed. The first `node_modules/.bin` entry on the script's PATH + (the package's own, then each parent directory's) must be the binary of the tool's own + installed package, and a symlink must resolve into it. A copy of a package manager or a + system program there is a shim. A same-named executable in the package root (`npm.cmd`) + counts too, because Windows runs it first. yarn must be a release: a `yarnPath` outside + `.yarn/releases/`, or a `packageManager` fetched from a URL, is refused. - Configuration that narrows the run is honoured: `.npmrc` or environment - `workspace`/`filter`/`script-shell`, lerna `command.run` filters, `.nxignore`, a - redefined nx `test` target, and turbo per-package tasks. + `workspace`/`filter`/`script-shell`, a `node-options` that loads code, lerna `command.run` + filters, `.nxignore`, a redefined nx `test` target or nx plugins (which can define it), and + turbo per-package tasks. - A cached result is not a run: turbo needs `--force` (or `cache: false` for `test`), and nx and lerna need `--skip-nx-cache`. - Only members of that tool's own workspace list count, read for the package manager in use. @@ -718,11 +731,11 @@ conservative too: The `coverage basis` line says what each package's verdict rests on: `measured` (its own suite ran), `inferred` (the recognized root command) or `declared` (`workspaces: "root"`). A -recognized root command that was not credited is printed as `root run not credited`, with the -reason. forge trusts the repository's own tooling: a shim that replaces the package manager -or the test runner is outside what it checks (one in `node_modules/.bin` shadowing -npm/pnpm/yarn is refused). A `script-shell` that is not a shell (`/bin/true`) makes npm and -pnpm suites `INCOMPLETE`. Fixture and test-data packages are never required. A test runner +root command that was not credited, including one that names a program by its path, is +printed as `root run not credited`, with the reason. Beyond these checks forge trusts the +installed tools themselves: it does not audit what a genuine package's binary or the test +runner does. A `script-shell` that is not a shell (`/bin/true`), or that is a program inside +the project, makes npm and pnpm suites `INCOMPLETE`. Fixture and test-data packages are never required. A test runner that is only a devDependency is not an obligation either; `forge stack` lists it as `available`. Tune this per repo under `verify` in `.forge/forge.config.json`: @@ -1170,10 +1183,20 @@ unchanged. never share a key. Artifacts minted before key version 3 never hit exact. Inline code the ledger's storage would rewrite (CRLF line endings, non-NFC text) is refused at mint; mint it from a file instead. -- **Near must also agree on behaviour.** A reworded match is offered as `near` only when the +- **Only exact establishes the same task.** Every hit carries `semanticEquivalence`: + `"identical"` for an exact hit, `"unverified"` for near and adapt. It also carries + `requiresReview`, `true` for every non-exact hit. A near hit is a similar task that no check + has shown to mean the same thing: review it against yours before reusing it. + `revalidation` answers a different question, namely whether the artifact and its + dependencies still hold. +- **Near must also agree on behaviour.** A similar spec is offered as `near` only when the two specs agree, in the same order, on these: - operators and symbols, with their operands (`x + 1` vs `x - 1`, `a - b` vs `b - a`); - numbers, identifiers, paths and negation; + - what each polarity, negation or direction word applies to. "Allow admins and deny + guests" vs "Deny admins and allow guests", or "from staging to production" vs "from + production to staging", conflict on `binding`, and so does the same set of words in a + different order; - literals, typographic and backtick quotes included (“a b”, ``a b``); - every whitespace run other than one space (line breaks, indentation, tabs, CRLF, columns); - spelling (`parse` vs `Parse`, fullwidth or Cyrillic look-alikes) and invisible format @@ -1203,6 +1226,7 @@ $ forge reuse query "debounce user input before firing search" sim: minhash NEAR hit (similarity 0.87) — module at src/lib/debounce.js claim 9c41d2ab77e0 — `forge ledger blame 9c41d2ab` for its proof + near tier: a similar task, equivalence unverified — review it against yours before reusing $ forge reuse query "quantum blockchain" sim: minhash diff --git a/docs/plans/substrate-v2/03-reuse-cache.md b/docs/plans/substrate-v2/03-reuse-cache.md index f9c73d8..5fe0e4a 100644 --- a/docs/plans/substrate-v2/03-reuse-cache.md +++ b/docs/plans/substrate-v2/03-reuse-cache.md @@ -48,7 +48,12 @@ neighbourhood" are different questions: exact tier now compares `keyHash`, a digest of the text's code units, and nothing is normalized at that boundary; similarity search stays layout-blind. The near tier compares the stored `key` with the semantic guard, so it requires a key of the current version - that the ledger stored verbatim (`keyVerbatim`: no CRLF or non-NFC text to fold). + that the ledger stored verbatim (`keyVerbatim`: no CRLF or non-NFC text to fold). The + guard's `binding` kind (review Q01) also compares what each polarity or direction word + applies to, so "Allow admins and deny guests" and "Deny admins and allow guests", which + share every token, are held at adapt. No token check establishes that two texts mean the + same, so every hit states what it establishes: `semanticEquivalence` is `"identical"` only + for exact and `"unverified"` for near and adapt, which also carry `requiresReview: true`. - **shape** (`spec`): identity plus typed placeholders for identifiers, paths, numbers and string literals (`⟨ident⟩`, `⟨path⟩`, `⟨num⟩`, `⟨str⟩`). diff --git a/mintlify/changelog/overview.mdx b/mintlify/changelog/overview.mdx index ec582b2..b89a381 100644 --- a/mintlify/changelog/overview.mdx +++ b/mintlify/changelog/overview.mdx @@ -18,6 +18,18 @@ This page is generated from `CHANGELOG.md` by `forge docs render`, and `forge do CI when it falls behind, so it cannot drift from the release notes again. {/* forge:render:changelog:begin (generated by `forge docs render` — do not edit) */} + + +**Fixed** + +- **A local program named like a package manager no longer covers the workspaces (Q02).** +- **A near reuse hit is no longer presented as the same task (Q01).** +- **The claims check no longer fails in a checkout that lacks newer release tags.** + +[Full notes for Unreleased →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#unreleased) + + + **Fixed** diff --git a/mintlify/cli/memory.mdx b/mintlify/cli/memory.mdx index 80394f9..5e26b4d 100644 --- a/mintlify/cli/memory.mdx +++ b/mintlify/cli/memory.mdx @@ -92,12 +92,18 @@ forge reuse stats # cache stats `"a b"`, indentation and Unicode code points all count, as do case, operators and punctuation, so `age >= 18` and `age <= 18` never share a key. Inline code that the ledger's storage would rewrite (CRLF, non-NFC text) is refused at mint: use `--file`. -- **Near must also agree on behaviour.** A reworded match drops to `adapt`, with a note +- **Only exact establishes the same task.** Every hit carries `semanticEquivalence` + (`"identical"` for exact, `"unverified"` for near and adapt) and `requiresReview` (`true` + for every non-exact hit). A near hit is a similar task, not a proven equivalent: review it + against yours before reusing it. +- **Near must also agree on behaviour.** A similar spec drops to `adapt`, with a note naming the difference, when anything behaviour-bearing differs, in content or in order: operators with their operands, numbers, literals (typographic quotes included), - identifiers, paths, negation, any whitespace other than one space, spelling (case, - look-alike letters) or invisible characters. A key from an older version, or one the - ledger stored normalized, never reaches near. + identifiers, paths, negation, what each polarity or direction word applies to ("allow + admins, deny guests" vs "deny admins, allow guests"; "from staging to production" vs the + reverse), any whitespace other than one space, spelling (case, look-alike letters) or + invisible characters. A key from an older version, or one the ledger stored normalized, + never reaches near. - **Its dependencies are what its imports bind to.** A named import records its defining module's declaration, followed through re-exports and export aliases. A default, namespace or side-effect import, `require` or `import()` records the whole module's diff --git a/mintlify/cli/quality.mdx b/mintlify/cli/quality.mdx index deb5987..0aa0a90 100644 --- a/mintlify/cli/quality.mdx +++ b/mintlify/cli/quality.mdx @@ -37,12 +37,20 @@ script covers the workspaces in one run only when forge establishes, from its sh structure, an unfiltered recursive run of every member's `test` script whose failure reaches the exit status (`npm test --workspaces`, `pnpm -r test`, `turbo run test`, …). Filters, masked runs and look-alike flags such as Node's preload `node -r` cover nothing, and each -package then runs its own suite. Nothing that could narrow or replay the run is credited: -options are allowlisted per tool, a script that changes shell state (`cd`, `exit`, `trap`) or -sets `npm_config_*` is refused, `.npmrc`/lerna/nx/turbo config that filters the run is -honoured, and turbo needs `--force` and nx/lerna `--skip-nx-cache` so a cached result is -never replayed. `coverage basis` labels every verdict `measured`, `inferred` or `declared`, -and a root command that was recognized but not credited is printed with the reason. Tune it +package then runs its own suite. A tool counts only by its bare name or its own +`node_modules/.bin` install: a program named by any other path (`./tools/npm`) is unknown, +whatever its name. Nothing that could narrow, replace or replay the run is credited: +- options are allowlisted per tool, and so are the variables the command sets (`PATH`, + `LD_PRELOAD` or `npm_config_*` are refused); +- a script that changes shell state (`cd`, `exit`, `trap`) is refused; +- a shim of the tool (a package manager in `node_modules/.bin`, another package's binary, a + same-named Windows executable in the root, a `yarnPath` outside `.yarn/releases/`) is + refused; +- `.npmrc`/lerna/nx/turbo config that filters the run is honoured; +- turbo needs `--force` and nx/lerna `--skip-nx-cache`, so a cached result is never replayed. + +`coverage basis` labels every verdict `measured`, `inferred` or `declared`, and a root +command that was not credited is printed with the reason. Tune it under `verify` in `.forge/forge.config.json`: ```json diff --git a/mintlify/concepts/proof-carrying-memory.mdx b/mintlify/concepts/proof-carrying-memory.mdx index 8ee2aaa..3ac7b5b 100644 --- a/mintlify/concepts/proof-carrying-memory.mdx +++ b/mintlify/concepts/proof-carrying-memory.mdx @@ -106,8 +106,12 @@ byte-identical to what was verified, and its dependencies must still resolve wit declarations. Otherwise it falls through to generation and mints a fresh claim on the way back. An exact hit needs the same text, byte for byte: no whitespace or Unicode normalization, because the spaces in `"a b"` or a Python block's indentation are data. A -near hit must also agree on operators, numbers, literals, identifiers, paths, negation and -code layout, so `age >= 18` is never served for `age <= 18`. +near hit must also agree on operators, numbers, literals, identifiers, paths, negation, what +each polarity or direction word applies to, and code layout, so `age >= 18` is never served +for `age <= 18`, nor "allow admins, deny guests" for "deny admins, allow guests". Only an +exact hit establishes the same task (`semanticEquivalence: "identical"`). A near or adapt +hit is a similar candidate whose equivalence is unverified, and it carries +`requiresReview: true`. One event is one vote. Abbreviations of one commit count once. A reworded lesson inherits trust only when the rewrite is equivalent. Similar-but-opposite rules are reported as diff --git a/mintlify/concepts/verification-gates.mdx b/mintlify/concepts/verification-gates.mdx index d3ef59f..af847da 100644 --- a/mintlify/concepts/verification-gates.mdx +++ b/mintlify/concepts/verification-gates.mdx @@ -64,12 +64,20 @@ script covers the workspaces in one run only when forge establishes, from its sh structure, an unfiltered recursive run of every member's `test` script whose failure reaches the exit status (`npm test --workspaces`, `pnpm -r test`, `turbo run test`, …). Filters, masked runs and look-alike flags such as Node's preload `node -r` cover nothing, and each -package then runs its own suite. Nothing that could narrow or replay the run is credited: -options are allowlisted per tool, a script that changes shell state (`cd`, `exit`, `trap`) or -sets `npm_config_*` is refused, `.npmrc`/lerna/nx/turbo config that filters the run is -honoured, and turbo needs `--force` and nx/lerna `--skip-nx-cache` so a cached result is -never replayed. `coverage basis` labels every verdict `measured`, `inferred` or `declared`, -and a root command that was recognized but not credited is printed with the reason. Tune it +package then runs its own suite. A tool counts only by its bare name or its own +`node_modules/.bin` install: a program named by any other path (`./tools/npm`) is unknown, +whatever its name. Nothing that could narrow, replace or replay the run is credited: +- options are allowlisted per tool, and so are the variables the command sets (`PATH`, + `LD_PRELOAD` or `npm_config_*` are refused); +- a script that changes shell state (`cd`, `exit`, `trap`) is refused; +- a shim of the tool (a package manager in `node_modules/.bin`, another package's binary, a + same-named Windows executable in the root, a `yarnPath` outside `.yarn/releases/`) is + refused; +- `.npmrc`/lerna/nx/turbo config that filters the run is honoured; +- turbo needs `--force` and nx/lerna `--skip-nx-cache`, so a cached result is never replayed. + +`coverage basis` labels every verdict `measured`, `inferred` or `declared`, and a root +command that was not credited is printed with the reason. Tune it under `verify` in `.forge/forge.config.json`: ```json diff --git a/scripts/claims-status.mjs b/scripts/claims-status.mjs index ddc597c..c3f68a4 100644 --- a/scripts/claims-status.mjs +++ b/scripts/claims-status.mjs @@ -190,17 +190,33 @@ export function validateRegistry(registry, { root = null } = {}) { return errors; } +/** Compare two `x.y.z` versions numerically (a pre-release suffix is ignored). */ +export function compareVersions(a, b) { + const parts = (v) => + String(v) + .split(/[.+-]/) + .slice(0, 3) + .map((n) => Number.parseInt(n, 10) || 0); + const [x, y] = [parts(a), parts(b)]; + for (let i = 0; i < 3; i++) if (x[i] !== y[i]) return x[i] - y[i]; + return 0; +} + /** * Cross-check each claim's `assessed_release` against the checkout's release tags: a claim * marked "unreleased" whose source commit has since shipped is stale (record the release), and * a claim naming a release must name one whose tag exists and contains its source commit. * Returns null when the check cannot run — no git, or no `v*` tags (a shallow CI clone) — and - * skips a claim whose commit is not in this clone. + * skips a claim whose commit is not in this clone. A checkout can also carry only SOME tags (a + * branch fetched without `--tags`): a release newer than every tag present but not newer than + * the code's own package.json version cannot be checked there — it is listed in `unchecked` + * (fetch the tags to check it), not reported as a problem. * @param {string} root * @param {any} registry a valid registry + * @param {{unchecked?: string[]}} [opts] receives the claims that could not be checked * @returns {string[]|null} */ -export function releaseProblems(root, registry) { +export function releaseProblems(root, registry, { unchecked = [] } = {}) { const g = (args) => execFileSync("git", args, { cwd: root, @@ -222,6 +238,14 @@ export function releaseProblems(root, registry) { return null; } if (!tags.size) return null; + const newest = [...tags] + .map((t) => t.slice(1)) + .sort(compareVersions) + .at(-1); + let version = null; + try { + version = JSON.parse(readFileSync(path.join(root, "package.json"), "utf8")).version ?? null; + } catch {} const firstRelease = new Map(); const shippedIn = (commit) => { if (!firstRelease.has(commit)) { @@ -247,9 +271,15 @@ export function releaseProblems(root, registry) { `${c.id}: assessed_release is "unreleased", but its source commit ${at} shipped in ${first} — record the release`, ); } else if (!tags.has(`v${c.assessed_release}`)) { - problems.push( - `${c.id}: assessed_release ${c.assessed_release} has no v${c.assessed_release} tag`, - ); + const untagged = + typeof version === "string" && + compareVersions(c.assessed_release, newest) > 0 && + compareVersions(c.assessed_release, version) <= 0; + if (untagged) unchecked.push(`${c.id} (${c.assessed_release})`); + else + problems.push( + `${c.id}: assessed_release ${c.assessed_release} has no v${c.assessed_release} tag`, + ); } else if (!ok(["merge-base", "--is-ancestor", c.source_commit, `v${c.assessed_release}`])) { problems.push( `${c.id}: release ${c.assessed_release} does not contain its source commit ${at}`, @@ -421,12 +451,18 @@ export function run(argv, io = {}) { for (const p of problems) error(`registry: ${p}`); return 1; } - const releases = releaseProblems(root, registry); + /** @type {string[]} */ + const unchecked = []; + const releases = releaseProblems(root, registry, { unchecked }); if (releases === null) log("release check skipped: no v* tags in this checkout"); else if (releases.length) { for (const p of releases) error(`registry: ${p}`); return 1; } + if (unchecked.length) + log( + `release check incomplete: ${unchecked.length} claim(s) name a release newer than this checkout's tags — \`git fetch --tags\` to check ${unchecked.join(", ")}`, + ); let failed = false; const readmeFile = path.join(root, README_PATH); diff --git a/src/cli/memory.js b/src/cli/memory.js index 1241e4e..cd76173 100644 --- a/src/cli/memory.js +++ b/src/cli/memory.js @@ -502,6 +502,9 @@ HANDLERS.reuse = async (argv) => { sim: r.sim, revalidation: r.revalidation?.status, requiresRevalidation: r.requiresRevalidation === true, + ...(r.tier === "miss" + ? {} + : { semanticEquivalence: r.semanticEquivalence, requiresReview: r.requiresReview }), reasons: r.reasons, }, null, @@ -520,9 +523,13 @@ HANDLERS.reuse = async (argv) => { ` claim ${a.id.slice(0, 12)} — \`forge ledger blame ${a.id.slice(0, 8)}\` for its proof`, ); if (r.tier === "near") - console.log(" near tier: a reworded match — review the diff before reusing it as-is"); + console.log( + " near tier: a similar task, equivalence unverified — review it against yours before reusing", + ); if (r.tier === "adapt") - console.log(" adapt tier: inject as a verified starting point, generate only the delta"); + console.log( + " adapt tier: a verified starting point for a similar task — generate only the delta, and review it", + ); if (r.requiresRevalidation) console.log( ` NOT revalidated: ${(r.revalidation?.unknown ?? []).join(", ")} — check before use`, diff --git a/src/reuse.js b/src/reuse.js index 68731e0..502ef2d 100644 --- a/src/reuse.js +++ b/src/reuse.js @@ -711,8 +711,11 @@ export function revalidate(artifact, atlas, { root = null } = {}) { * null for that candidate (missing vector), MinHash Jaccard with NEAR_J/ADAPT_J is the * per-candidate fallback — a partially-embedded ledger never loses lexical recall. * Similarity alone never serves code as-is (review F04): a near candidate whose operators, - * numbers, literals, identifiers, paths or polarity words differ from the query is held at - * the adapt tier (a starting point to review), with the conflict named in `reasons`. + * numbers, literals, identifiers, paths, polarity words or their bindings differ from the + * query is held at the adapt tier (a starting point to review), with the conflict named in + * `reasons`. And similarity never establishes equivalence (review Q01): every hit carries + * `semanticEquivalence` — "identical" for exact (the byte-identical spec), "unverified" for + * near and adapt — and `requiresReview`, true for every non-exact hit. * Every hit is revalidated at the serving boundary (review F05, see revalidate): an * `invalid` artifact is never served; an `unknown` one is returned with * `requiresRevalidation: true` — never presented as checked. @@ -722,6 +725,7 @@ export function revalidate(artifact, atlas, { root = null } = {}) { * sim?:((query:any, claim:any)=>number|null)|null}} opts * @returns {{tier:"exact"|"near"|"adapt"|"miss", artifact?:any, jaccard?:number, * similarity?:number, simBackend?:string, revalidation?:object, + * semanticEquivalence?:"identical"|"unverified", requiresReview?:boolean, * requiresRevalidation?:boolean, reasons:string[], sim?:string, * invalidated?:{id:string, missing:string[], changed:string[], problems:string[]}[]}} * `sim` is stamped by reuseQuery/reusePeek (the backend label the CLI prints); @@ -758,10 +762,19 @@ export function lookup( reasons.push(`${why} ${c.id.slice(0, 8)} failed revalidation: ${rv.problems.join("; ")}`); return null; }; + // Review Q01: only the exact tier ESTABLISHES that the task is the one the artifact was + // verified for (the same spec, byte for byte). A near or adapt hit is a similar candidate — + // no check here can establish that two texts mean the same — so it is always marked as one + // that requires review, with its equivalence unverified. `revalidation` is a separate + // question: whether the artifact and its dependencies still hold, not whether it fits. const served = (tier, c, rv, extra = {}) => ({ tier, artifact: c, ...extra, + semanticEquivalence: /** @type {"identical"|"unverified"} */ ( + tier === "exact" ? "identical" : "unverified" + ), + requiresReview: tier !== "exact", revalidation: rv, ...(rv.status === "valid" ? {} : { requiresRevalidation: true }), reasons, @@ -796,7 +809,7 @@ export function lookup( const qBands = new Set(bandKeys(qs)); pool = artifacts.filter((c) => bandKeys(shapeOf(c)).some((k) => qBands.has(k))); } - // near compares IDENTITY (same names, reworded prose); adapt compares SHAPE too, so + // near compares IDENTITY (same names, similar prose); adapt compares SHAPE too, so // `add pagination to listOrders` can still be offered the listUsers artifact as a // starting point — the tier that says "generate only the delta" — but never as-is. const measure = (c) => { diff --git a/src/semantic_guard.js b/src/semantic_guard.js index b850f00..a69266f 100644 --- a/src/semantic_guard.js +++ b/src/semantic_guard.js @@ -430,13 +430,159 @@ function spellingsOf(masked) { return map; } +// --------------------------------------------------------------------------------------- +// Relational binding (review Q01). "Allow admins and deny guests…" and "Deny admins and allow +// guests…" hold the same tokens — the same polarity words, the same names — and score as +// near-identical; what differs is WHICH word each polarity or direction word applies to. Each +// relation word (a negator, a pole of an antonym pair, a direction preposition) is bound to +// the next content word of its clause, a quoted literal included. Two texts conflict on +// `binding` when +// 1. a word both texts bind is bound to OPPOSITE relations (allow→admins vs deny→admins, +// from→staging vs to→staging; a negator bound to the same word flips its poles), or +// 2. they use exactly the same words in a different order — a rearrangement, which can swap +// who is allowed, or source and destination, without changing a single token. +// A token heuristic, not a parser: it holds known reversals back from the near tier. It +// cannot establish that two texts mean the same — nothing in this module can — which is why +// every non-exact reuse hit is a candidate that requires review (reuse.js), never an +// equivalent. +// --------------------------------------------------------------------------------------- + +/** word → every [relation class, pole] it expresses (antonym pairs, then direction). */ +const RELATION_POLES = new Map(); +/** @param {Iterable} words @param {string} cls @param {number} pole */ +const addPoles = (words, cls, pole) => { + for (const w of words) RELATION_POLES.set(w, [...(RELATION_POLES.get(w) ?? []), [cls, pole]]); +}; +ANTONYM_PAIRS.forEach(([a, b], i) => { + addPoles(a, `p${i}`, 0); + addPoles(b, `p${i}`, 1); +}); +addPoles(["from"], "direction", 0); +addPoles(["to", "into", "onto", "toward", "towards"], "direction", 1); +// Function words passed over when looking for the word a relation applies to. +const BIND_SKIP = new Set( + ( + "a an the all any every each some only just also both either its it their them they our " + + "your his her my this that these those same other such be is are was were been being am " + + "do does did have has had will would shall should can could might of for with by at in as " + + "per via about up down out" + ).split(" "), +); +// Words that end the phrase a pending relation can apply to. +const CLAUSE_WORDS = new Set( + "and or but then while whereas unless except if when so yet".split(" "), +); +const isNegator = (w) => NEGATORS.has(w) || NEGATED_CONTRACTION.test(w); +const CLAUSE_END = new Set([...",;:.!?"]); +const CLOSING = new Set([..."\"'”’)]"]); + +/** + * The words of a masked text (lower-cased, a possessive `'s` dropped; a literal is its own + * text) and each relation + * word's binding `rel→object`, both in document order. + * @param {string} masked + * @param {string[]} literals + * @returns {{words: string[], bindings: {rel: string, obj: string}[]}} + */ +function relationsOf(masked, literals) { + /** @type {string[]} */ + const words = []; + /** @type {{rel: string, obj: string}[]} */ + const bindings = []; + /** @type {string[]} */ + let pending = []; + let lit = 0; + for (const raw of masked.split(/\s+/u)) { + if (!raw) continue; + const marks = [...raw].filter((ch) => ch === LIT).length; + const quoted = marks ? literals.slice(lit, lit + marks).join(" ") : ""; + lit += marks; + // a possessive names the same party: "the author's change" binds `author` + const word = + quoted || + trimEdges(raw.replaceAll(FENCE, "")) + .toLowerCase() + .replace(/['’]s$/u, ""); + if (!word || !/[\p{L}\p{N}]/u.test(word)) { + pending = []; // a dash, a bare symbol: the phrase ends + continue; + } + words.push(word); + if (!quoted && (isNegator(word) || RELATION_POLES.has(word))) pending.push(word); + else if (!quoted && CLAUSE_WORDS.has(word)) pending = []; + else if (quoted || !BIND_SKIP.has(word)) { + for (const rel of pending) bindings.push({ rel, obj: word }); + pending = []; + } + let end = raw.length; + while (end > 0 && CLOSING.has(raw[end - 1])) end -= 1; + if (end > 0 && CLAUSE_END.has(raw[end - 1])) pending = []; // `admins,` ends the phrase + } + return { words, bindings }; +} + +/** Each bound word's relation classes, with the pole a negator on the same word flips. */ +function relationSignatures(bindings) { + /** @type {Map} */ + const byObj = new Map(); + for (const { rel, obj } of bindings) { + const rels = byObj.get(obj); + if (rels) rels.push(rel); + else byObj.set(obj, [rel]); + } + /** @type {Map>} */ + const out = new Map(); + for (const [obj, rels] of byObj) { + const flip = rels.filter(isNegator).length % 2; + const sig = new Set(); + for (const rel of rels) + for (const [cls, pole] of RELATION_POLES.get(rel) ?? []) sig.add(`${cls}:${pole ^ flip}`); + out.set(obj, sig); + } + return out; +} + +/** + * The `binding` conflict between two texts' relations (see above), or null. + * @param {{words: string[], bindings: {rel: string, obj: string}[]}} A + * @param {{words: string[], bindings: {rel: string, obj: string}[]}} B + * @returns {{kind: string, a: string[], b: string[], order?: boolean}|null} + */ +function bindingConflict(A, B) { + const shown = (bs) => bs.map((x) => `${x.rel}→${x.obj}`); + // 1. a word both bind, bound to opposite poles of one relation + const [sa, sb] = [relationSignatures(A.bindings), relationSignatures(B.bindings)]; + const opposed = new Set(); + const opposite = (x, y) => + [...x].some((k) => !y.has(k) && y.has(`${k.slice(0, -1)}${1 - Number(k.slice(-1))}`)); + for (const [obj, x] of sa) { + const y = sb.get(obj); + if (y && (opposite(x, y) || opposite(y, x))) opposed.add(obj); + } + if (opposed.size) + return { + kind: "binding", + a: shown(A.bindings.filter((x) => opposed.has(x.obj))), + b: shown(B.bindings.filter((x) => opposed.has(x.obj))), + }; + // 2. the same words, rearranged + if (sameList(A.words, B.words) || !sameList(sorted(A.words), sorted(B.words))) return null; + let i = 0; + while (A.words[i] === B.words[i]) i += 1; + let j = 0; + while (A.words[A.words.length - 1 - j] === B.words[B.words.length - 1 - j]) j += 1; + const span = (ws) => ws.slice(i, ws.length - j); + const clip = (ws) => (ws.length > 12 ? [...ws.slice(0, 12), "…"] : ws); + return { kind: "binding", a: clip(span(A.words)), b: clip(span(B.words)), order: true }; +} + /** * The behaviour-carrying features of a text, each in DOCUMENT ORDER (review N01 round 2: * sorted lists made `a - b` and `b - a`, or two swapped indented lines, the same text). * @param {string} text * @returns {{operators: string[], numbers: string[], literals: string[], - * identifiers: string[], paths: string[], polarity: string[], layout: string[], - * symbols: string[], format: string[]}} + * identifiers: string[], paths: string[], polarity: string[], bindings: string[], + * layout: string[], symbols: string[], format: string[]}} */ export function criticalFeatures(text) { return analyze(text).features; @@ -478,6 +624,7 @@ function analyze(text) { `${cf.map((ch) => `U+${(ch.codePointAt(0) ?? 0).toString(16).toUpperCase().padStart(4, "0")}`).join(" ")} in ${JSON.stringify(tok)}`, ); } + const relations = relationsOf(masked, literals); const features = { operators, numbers, @@ -485,11 +632,12 @@ function analyze(text) { identifiers, paths, polarity: sorted(polarity), + bindings: relations.bindings.map((x) => `${x.rel}→${x.obj}`), layout: [...fences.map((f) => `fence ${JSON.stringify(f)}`), ...whitespaceItems(masked)], symbols: symbolRuns(masked), format, }; - return { features, masked }; + return { features, masked, relations }; } const sameList = (a, b) => a.length === b.length && a.every((x, i) => x === b[i]); @@ -517,6 +665,7 @@ export function polarityFlip(a, b) { * a rewrite that names a new identifier or path is more specific, not the opposite). */ export const ALL_KINDS = /** @type {const} */ ([ "polarity", + "binding", "operators", "numbers", "literals", @@ -529,6 +678,7 @@ export const ALL_KINDS = /** @type {const} */ ([ ]); export const FLIP_KINDS = /** @type {const} */ ([ "polarity", + "binding", "operators", "numbers", "literals", @@ -551,6 +701,11 @@ export function semanticConflicts(a, b, { kinds = ALL_KINDS } = {}) { const [fa, fb] = [A.features, B.features]; const out = []; for (const kind of kinds) { + if (kind === "binding") { + const c = bindingConflict(A.relations, B.relations); + if (c) out.push(c); + continue; + } if (kind === "spelling") { const sa = spellingsOf(A.masked); const sb = spellingsOf(B.masked); diff --git a/src/stack.js b/src/stack.js index fd5fb40..e8a5098 100644 --- a/src/stack.js +++ b/src/stack.js @@ -4,8 +4,8 @@ // Everything is data: SIGNATURES maps a dependency/marker to a label, so widening coverage // is adding a row, never editing logic. Fail-safe — an unreadable or absent manifest is // skipped, never thrown. -import { existsSync, readdirSync, readFileSync } from "node:fs"; -import { join } from "node:path"; +import { existsSync, lstatSync, readdirSync, readFileSync, realpathSync } from "node:fs"; +import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path"; import { stripTrailingSlashes } from "./util.js"; const read = (root, rel) => { @@ -622,7 +622,42 @@ const CACHE_BYPASS_ENV = new Map([ ["TURBO_FORCE", "turbo"], ["NX_SKIP_NX_CACHE", "nx"], ]); +// Variables a recognized command line may set — an ALLOWLIST, like the options (review Q02). +// Any other can change which program a name runs (PATH), what it loads (LD_PRELOAD, +// NODE_OPTIONS=--require), where it reads its configuration (HOME), or which members it runs. +const INERT_ENV = new Set([ + "CI", + "NODE_ENV", + "FORCE_COLOR", + "NO_COLOR", + "TZ", + "LANG", + "LC_ALL", + "DEBUG", + "DO_NOT_TRACK", + "NX_DAEMON", + "NX_NO_CLOUD", + "TURBO_TELEMETRY_DISABLED", +]); +// NODE_OPTIONS flags that neither load code nor soften a failure. +const INERT_NODE_OPTION = + /^--(?:(?:max[-_]old[-_]space[-_]size|max[-_]semi[-_]space[-_]size|stack[-_]trace[-_]limit)=\d+|experimental-vm-modules|no-warnings|no-deprecation|trace-warnings|trace-deprecation|trace-uncaught|enable-source-maps|unhandled-rejections=(?:strict|throw))$/; const TRUTHY = new Set(["1", "true"]); + +/** Whether a NODE_OPTIONS value holds only inert flags. */ +const inertNodeOptions = (value) => + String(value) + .split(/\s+/) + .every((o) => !o || INERT_NODE_OPTION.test(o)); + +/** Why a variable set on the command line keeps a run from being established, or null. */ +function envProblem(name, value) { + if (INERT_ENV.has(name) || CACHE_BYPASS_ENV.has(name)) return null; + if (name === "NODE_OPTIONS" && inertNodeOptions(value)) return null; + if (/^path(?:ext)?$/i.test(name)) return `sets ${name}, which changes the program a name runs`; + if (TOOL_ENV.test(name)) return `sets ${name}, which can narrow or redirect the run`; + return `sets ${name}, which is not a variable known to leave the run unchanged`; +} // Shell builtins that change what LATER commands do — the directory (`cd packages/good && npm // test --workspaces` runs one workspace), the exit status (`trap 'exit 0' EXIT`, `… && exit // 0; …`), the environment, the meaning of a name. A script holding any is not a recognized @@ -635,6 +670,8 @@ const STATEFUL_BUILTINS = new Set( ); // Wrappers whose job is to find and run another binary. const TOOL_BINS = new Set(["turbo", "lerna", "nx"]); +// npx/bunx/npm exec options that change nothing about which binary runs. `-p`/`--package` +// is not one: it picks the package that provides the binary (review Q02). const NPX_OK = new Set([ "--yes", "-y", @@ -644,21 +681,65 @@ const NPX_OK = new Set([ "-q", "--prefer-offline", ]); -/** `./node_modules/.bin/turbo.cmd` → `turbo`. */ -const binName = (w) => +// Programs a project installs as a dependency, and the package whose binary each must be. A +// bare name finds them through the script's PATH (node_modules/.bin first); the same install +// may be spelled out as `node_modules/.bin/`. Package managers and system programs are +// never project binaries (runProblem refuses a copy in node_modules/.bin). +const INSTALLED_BINS = new Map([ + ["turbo", "turbo"], + ["nx", "nx"], + ["lerna", "lerna"], + ["cross-env", "cross-env"], + ["bun", "bun"], + ["bunx", "bun"], +]); + +/** + * The program a command word names, as far as recognition may rely on it (review Q02). A bare + * name is the program of that name the script's PATH resolves — runProblem checks that nothing + * in the project shadows it. `node_modules/.bin/` (optionally `./`-prefixed, or its + * Windows `.cmd`/`.ps1` shim) is that same installed tool spelled out. Any other path — + * `./tools/npm`, `./scripts/pnpm.js`, `/usr/bin/env` — is a program whose behaviour is + * unknown, whatever its basename: null. + * @param {string} word + * @returns {string|null} + */ +function commandIdentity(word) { + const w = String(word); + if (!/[\\/]/.test(w)) return w; + const m = /^(?:\.[\\/])?node_modules[\\/]\.bin[\\/]([^\\/]+?)(?:\.cmd|\.ps1)?$/i.exec(w); + return m && INSTALLED_BINS.has(m[1]) ? m[1] : null; +} + +/** The name a word's basename suggests (`./tools/npm.cmd` → `npm`) — used only to explain a + * refusal, never to recognize a program. */ +const suggestedName = (w) => String(w) .split(/[\\/]/) .pop() - .replace(/\.(?:cmd|exe|ps1|js|cjs|mjs)$/i, ""); + ?.replace(/\.(?:cmd|exe|ps1|bat|js|cjs|mjs)$/i, "") ?? ""; -/** Strip env assignments and runner wrappers (env, cross-env, npx, bunx, npm exec, pnpm/yarn - * exec|dlx, `yarn turbo`) down to the executable that does the work, with the variables the - * command line sets. `null` when a wrapper is used in a way that could change what runs - * (`npx -c`, `npm exec --workspaces`…) or a tool-namespaced variable is set. */ -function unwrapCommand(words) { +/** + * Strip env assignments and runner wrappers (env, cross-env, npx, bunx, npm exec, pnpm/yarn + * exec|dlx, `yarn turbo`) down to the program that does the work: its identity (`bin`), its + * arguments, the variables the command line sets, and `bins` — every program the line relies + * on, wrappers first, for runProblem's shadow checks. `null` when a wrapper is used in a way + * that could change what runs (`npx -c`, `npx -p pkg`, `env -C dir`, `npm exec + * --workspaces`…), a word names an unknown program (commandIdentity), or a variable outside + * the allowlist is set. `explain` relaxes the last two and records the first such reason as + * `note` instead — for a refusal message only. + * @param {string[]} words + * @param {boolean} [explain] + * @returns {{bin: string, args: string[], words: string[], env: Map, bins: string[], note: string|null}|null} + */ +function unwrapCommand(words, explain = false) { const w = [...words]; /** @type {Map} */ const env = new Map(); + /** @type {string[]} */ + const bins = []; + /** @type {string|null} */ + let note = null; const assign = () => { while (w.length && ENV_ASSIGN.test(w[0])) { const word = /** @type {string} */ (w.shift()); @@ -666,16 +747,29 @@ function unwrapCommand(words) { env.set(word.slice(0, eq), word.slice(eq + 1)); } }; + /** @param {string} word */ + const identify = (word) => { + const id = commandIdentity(word); + if (id !== null || !explain) return id; + const name = suggestedName(word); + note ??= `\`${word}\` is a program named by its path, not ${name} itself: only a bare name or a node_modules/.bin install is recognized`; + return name; + }; for (let hop = 0; hop < 6; hop++) { assign(); if (!w.length || w[0] === "!") return null; // `! cmd` inverts the status - const bin = binName(w[0]); + const bin = identify(w[0]); + if (bin === null) return null; + bins.push(bin); const next = w[1]; if (bin === "env" || bin === "cross-env") { w.shift(); + // `env -u NAME` only removes a variable; `-i`, `-C`/`--chdir`, `-S`… change the + // environment or the directory the command runs in. cross-env takes no options. while (w.length && w[0].startsWith("-")) { - if (w[0] === "-u" || w[0] === "--unset") w.shift(); - w.shift(); + if (bin === "env" && (w[0] === "-u" || w[0] === "--unset") && w.length > 1) w.splice(0, 2); + else if (bin === "env" && w[0].startsWith("--unset=")) w.shift(); + else return null; } continue; } @@ -684,9 +778,8 @@ function unwrapCommand(words) { if (bin === "bun") w.shift(); while (w.length && w[0].startsWith("-")) { const o = /** @type {string} */ (w.shift()); - if (o === "-p" || o === "--package") w.shift(); - else if (o === "--") break; - else if (!NPX_OK.has(o) && !o.startsWith("--package=")) return null; + if (o === "--") break; + if (!NPX_OK.has(o)) return null; } continue; } @@ -695,7 +788,7 @@ function unwrapCommand(words) { while (w.length && w[0].startsWith("-")) { const o = /** @type {string} */ (w.shift()); if (o === "--") break; - if (!NPX_OK.has(o) && !o.startsWith("--package=")) return null; + if (!NPX_OK.has(o)) return null; } continue; } @@ -705,13 +798,22 @@ function unwrapCommand(words) { if (w[0]?.startsWith("-")) return null; continue; } - if ((bin === "pnpm" || bin === "yarn") && next && TOOL_BINS.has(binName(next))) { + // `yarn turbo run test`: the package manager runs the installed tool, named bare. + if ( + (bin === "pnpm" || bin === "yarn") && + next !== undefined && + TOOL_BINS.has(explain ? suggestedName(next) : next) + ) { w.shift(); continue; } - for (const name of env.keys()) - if (TOOL_ENV.test(name) && !CACHE_BYPASS_ENV.has(name)) return null; - return { bin, args: w.slice(1), words: w, env }; + for (const [name, value] of env) { + const why = envProblem(name, value); + if (!why) continue; + if (!explain) return null; + note ??= `the command line ${why}`; + } + return { bin, args: w.slice(1), words: w, env, bins, note }; } return null; } @@ -933,37 +1035,71 @@ const MAX_SCRIPT_HOPS = 4; const firstWord = (c) => c.words.find((w) => !ENV_ASSIGN.test(w)) ?? ""; /** - * The recursive workspace test run a root `test` script performs, if one is ESTABLISHED: - * `{tool, command, bypass}` for the recognized invocation (`command` is its words, re-joined; - * `bypass` — the command itself disables the tool's result cache: `turbo --force`, - * `--skip-nx-cache`, `TURBO_FORCE=1`), else null. Pure — `scripts` is the root package.json's - * `scripts`, used to follow `npm run