From b7751f63dc83e7109dc0d484fd50e796ed44545b Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 21:31:18 +0000 Subject: [PATCH 1/4] fix(security): escape claim-table backslashes; remove quadratic regexes CodeQL flagged one new high-severity alert in #164 and two older ones. A local run of the same CodeQL version (2.27.0, javascript-code-scanning suite) goes from 3 results to 0 with this change. - scripts/claims-status.mjs (js/incomplete-sanitization, new in #164): table cells escaped `|` but not `\`, so a trailing backslash could undo the escape. Backslashes are escaped first. - src/model_catalog.js (js/polynomial-redos, pre-existing): the tokenizer's `/(?:\.0)+$/` and trimUrl's `/\/+$/` backtracked quadratically on library input. Trailing ".0" parts are popped instead, and URLs use a new linear stripTrailingSlashes (src/util.js). - The same patterns in code added by #164 are hardened too: the workspace-glob trim (src/stack.js) and the semantic guard's edge punctuation trim, now a code-point scan (trimEdges). Each replacement is checked against the regex it replaced (tokenize on known ids; trimEdges on 2,000 seeded random tokens) and gets a linear-time test on a hostile input. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- scripts/claims-status.mjs | 6 +++-- src/model_catalog.js | 8 ++++-- src/semantic_guard.js | 16 +++++++++++- src/stack.js | 5 ++-- src/util.js | 9 +++++++ test/claims_status.test.js | 8 ++++++ test/model_catalog.test.js | 50 +++++++++++++++++++++++++++++++++++++ test/semantic_guard.test.js | 41 ++++++++++++++++++++++++++++++ test/util.test.js | 12 ++++++++- 9 files changed, 147 insertions(+), 8 deletions(-) diff --git a/scripts/claims-status.mjs b/scripts/claims-status.mjs index 24d0c8c1..b911ae55 100644 --- a/scripts/claims-status.mjs +++ b/scripts/claims-status.mjs @@ -144,10 +144,12 @@ export function validateRegistry(registry, { root = null } = {}) { return errors; } -/** Escape a value for a single Markdown table cell. */ -const cell = (s) => +/** Escape a value for a single Markdown table cell: backslashes first, then the pipes a + * cell cannot contain (escaping only the pipes would let a trailing `\` undo the escape). */ +export const cell = (s) => String(s ?? "") .replace(/\r?\n/g, " ") + .replace(/\\/g, "\\\\") .replace(/\|/g, "\\|") .trim(); diff --git a/src/model_catalog.js b/src/model_catalog.js index e340e7cd..97841d2a 100644 --- a/src/model_catalog.js +++ b/src/model_catalog.js @@ -16,6 +16,7 @@ // function here is total: an unavailable catalog is `null`, never a throw. import { join } from "node:path"; import { cachedGetJson, httpGet } from "./http_cache.js"; +import { stripTrailingSlashes } from "./util.js"; export const ANTHROPIC_API = "https://api.anthropic.com"; export const ANTHROPIC_VERSION = "2023-06-01"; @@ -53,7 +54,10 @@ export function tokenize(s) { } const run = []; while (i < parts.length && isVersionPart(parts[i])) run.push(parts[i++]); - out.add(run.join(".").replace(/(?:\.0)+$/, "")); + // Drop trailing ".0" parts ("4.0" → "4") by popping: `/(?:\.0)+$/` backtracks + // quadratically on a long ".0.0.0…" run (polynomial ReDoS on library input). + while (run.length > 1 && run[run.length - 1] === "0") run.pop(); + out.add(run.join(".")); } return out; } @@ -268,7 +272,7 @@ export function matchCatalogModel(modelId, models) { * @typedef {{kind:null, reason:string}} NoCatalog */ -const trimUrl = (u) => String(u ?? "").replace(/\/+$/, ""); +const trimUrl = (u) => stripTrailingSlashes(u); /** The Anthropic Models API, authenticated with an API key (x-api-key). */ export function anthropicSource(apiKey) { diff --git a/src/semantic_guard.js b/src/semantic_guard.js index 06e9efad..404a3a1a 100644 --- a/src/semantic_guard.js +++ b/src/semantic_guard.js @@ -81,6 +81,20 @@ const PATH_RE = /[\\/]|\.(?:m?[jt]sx?|py|go|rs|java|rb|json|ya?ml|toml|md|css|ht const sorted = (xs) => [...xs].sort(); +// What may stay at a token's edges: a leading "." keeps dotfiles and relative paths, a +// trailing one is sentence punctuation. A code-point scan, not `/…|[^…]+$/g`, which +// backtracks quadratically on a long run of punctuation inside one token. +const KEEP_HEAD = /[\p{L}\p{N}_$./\\]/u; +const KEEP_TAIL = /[\p{L}\p{N}_$/\\]/u; +export function trimEdges(raw) { + const cps = [...raw]; + let a = 0; + let b = cps.length; + while (a < b && !KEEP_HEAD.test(cps[a])) a++; + while (b > a && !KEEP_TAIL.test(cps[b - 1])) b--; + return cps.slice(a, b).join(""); +} + /** * The behaviour-carrying features of a text. * @param {string} text @@ -103,7 +117,7 @@ export function criticalFeatures(text) { const polarity = []; for (const raw of rest.split(/\s+/)) { // edge punctuation only — inner punctuation (dots, underscores, slashes) is identity - const tok = raw.replace(/^[^\p{L}\p{N}_$./\\]+|[^\p{L}\p{N}_$/\\]+$/gu, ""); + const tok = trimEdges(raw); if (!tok) continue; if (PATH_RE.test(tok) && /[\p{L}\p{N}]/u.test(tok)) paths.push(tok); else if (IDENT_RE.test(tok)) identifiers.push(tok); diff --git a/src/stack.js b/src/stack.js index 1b141315..24531f44 100644 --- a/src/stack.js +++ b/src/stack.js @@ -6,6 +6,7 @@ // skipped, never thrown. import { existsSync, readdirSync, readFileSync } from "node:fs"; import { join } from "node:path"; +import { stripTrailingSlashes } from "./util.js"; const read = (root, rel) => { try { @@ -410,7 +411,7 @@ function scanPackageRoots(root) { * @param {string} glob @param {string} rel POSIX path relative to the repo root */ export function matchesWorkspaceGlob(glob, rel) { - const g = String(glob).replace(/^\.\//, "").replace(/\/+$/, ""); + const g = stripTrailingSlashes(String(glob).replace(/^\.\//, "")); let re = ""; for (let i = 0; i < g.length; i++) { const c = g[i]; @@ -423,7 +424,7 @@ export function matchesWorkspaceGlob(glob, rel) { else if (c === "?") re += "[^/]"; else re += c.replace(/[.+^${}()|[\]\\]/g, "\\$&"); } - return new RegExp(`^${re}$`).test(String(rel).replace(/\/+$/, "")); + return new RegExp(`^${re}$`).test(stripTrailingSlashes(rel)); } // A root test script that runs EVERY workspace's suite itself — so a nested package that is a diff --git a/src/util.js b/src/util.js index 6dad9cc7..4a7991ec 100644 --- a/src/util.js +++ b/src/util.js @@ -42,6 +42,15 @@ export const clamp01 = (x) => { // POSIX (no `\` in the path) and safe on Windows (fs/join accept `/`). export const toPosix = (p) => String(p).replaceAll("\\", "/"); +/** `s` without its trailing slashes. A backward scan, not `/\/+$/`: that regex backtracks + * quadratically on a long run of slashes followed by anything else (polynomial ReDoS). */ +export function stripTrailingSlashes(s) { + const t = String(s ?? ""); + let end = t.length; + while (end > 0 && t.charCodeAt(end - 1) === 47) end--; + return t.slice(0, end); +} + export const MS_PER_DAY = 86400000; export const epochDay = () => Math.floor(Date.now() / MS_PER_DAY); diff --git a/test/claims_status.test.js b/test/claims_status.test.js index 8ec60701..aabcf3a9 100644 --- a/test/claims_status.test.js +++ b/test/claims_status.test.js @@ -8,6 +8,7 @@ import { dirname, join } from "node:path"; import { test } from "node:test"; import { BEGIN, + cell, checkCopies, END, README_PATH, @@ -209,3 +210,10 @@ test("the repository's own registry is valid, its table current, and its docs co const r = cli(["--check"]); assert.equal(r.code, 0, r.err); }); + +test("table cells escape backslashes before pipes (a trailing backslash cannot undo an escape)", () => { + assert.equal(cell("a|b"), "a\\|b"); + assert.equal(cell("a\\|b"), "a\\\\\\|b"); + assert.equal(cell("ends with \\"), "ends with \\\\"); + assert.equal(cell("two\nlines"), "two lines"); +}); diff --git a/test/model_catalog.test.js b/test/model_catalog.test.js index fcd6dd27..1145ad39 100644 --- a/test/model_catalog.test.js +++ b/test/model_catalog.test.js @@ -11,12 +11,14 @@ import { canonicalKey, catalogSource, fetchCatalog, + gatewaySource, inFamily, matchCatalogModel, newestInFamily, normalizeCatalogPage, openRouterSource, perMillion, + tokenize, } from "../src/model_catalog.js"; import { anthropicPage, ok, stubTransport } from "./_catalog_stub.js"; @@ -273,3 +275,51 @@ test("fetchCatalog reports when its answer expires (the memo uses it, no invente const b = fetchCatalog(anthropicSource("sk-a"), { fetchImpl: noHeaders.fetchImpl, now: T }); assert.equal(b?.freshUntil, T, "no freshness stated → revalidate on the next use"); }); + +test("tokenize collapses trailing .0 versions exactly as before, in linear time", () => { + const old = (s) => { + // The previous implementation, kept here as the oracle (its regex was quadratic). + const parts = String(s ?? "") + .toLowerCase() + .split(/[^a-z0-9]+/) + .filter(Boolean); + const out = new Set(); + for (let i = 0; i < parts.length; ) { + if (!/^\d{1,3}$/.test(parts[i])) { + out.add(parts[i++]); + continue; + } + const run = []; + while (i < parts.length && /^\d{1,3}$/.test(parts[i])) run.push(parts[i++]); + out.add(run.join(".").replace(/(?:\.0)+$/, "")); + } + return out; + }; + for (const id of [ + "claude-sonnet-4-5-20250929", + "claude-opus-4-0", + "gpt-4.0", + "model-0-0", + "model-4-00", + "x-0", + "x-4-0-5", + "gemini-3-flash", + "a-10-0-0", + ]) + assert.deepEqual([...tokenize(id)], [...old(id)], id); + const hostile = `m-${"0-".repeat(60000)}1`; + const t0 = performance.now(); + tokenize(hostile); + assert.ok(performance.now() - t0 < 1000, "linear, not quadratic"); +}); + +test("catalog urls drop any run of trailing slashes, in linear time", () => { + assert.equal(gatewaySource("http://gw.local///").url, "http://gw.local/v1/models"); + assert.equal( + openRouterSource("https://openrouter.ai/api/v1/").url, + "https://openrouter.ai/api/v1/models", + ); + const t0 = performance.now(); + gatewaySource(`http://x${"/".repeat(200000)}a`); + assert.ok(performance.now() - t0 < 1000); +}); diff --git a/test/semantic_guard.test.js b/test/semantic_guard.test.js index 89550279..b2a93dfc 100644 --- a/test/semantic_guard.test.js +++ b/test/semantic_guard.test.js @@ -5,6 +5,7 @@ import { describeConflicts, sameSemantics, semanticConflicts, + trimEdges, } from "../src/semantic_guard.js"; // The review's four exact-key collisions (F04) and its opposite-rule pair (F16): similar text, @@ -68,3 +69,43 @@ test("semantic guard: features are extracted per class; literals are not re-scan polarity: [], }); }); + +test("trimEdges equals the edge-punctuation regex it replaced, in linear time", () => { + const OLD = /^[^\p{L}\p{N}_$./\\]+|[^\p{L}\p{N}_$/\\]+$/gu; // quadratic on long runs + const alphabet = [ + "a", + "Z", + "7", + "_", + "$", + ".", + "/", + "\\", + "!", + "?", + ",", + "(", + ")", + "é", + "ß", + "😀", + "-", + '"', + ]; + let seed = 20260926; + const rand = () => { + seed = (seed * 1103515245 + 12345) % 2147483648; + return seed / 2147483648; + }; + for (let trial = 0; trial < 2000; trial++) { + const n = Math.floor(rand() * 9); + let tok = ""; + for (let i = 0; i < n; i++) tok += alphabet[Math.floor(rand() * alphabet.length)]; + assert.equal(trimEdges(tok), tok.replace(OLD, ""), JSON.stringify(tok)); + } + // A long punctuation run inside one token: the old regex took quadratic time here. + const hostile = `a${"!".repeat(200000)}a`; + const t0 = performance.now(); + assert.equal(trimEdges(hostile), hostile); + assert.ok(performance.now() - t0 < 1000, "linear, not quadratic"); +}); diff --git a/test/util.test.js b/test/util.test.js index 136c78d1..3f67a334 100644 --- a/test/util.test.js +++ b/test/util.test.js @@ -1,6 +1,6 @@ import assert from "node:assert/strict"; import { test } from "node:test"; -import { clamp01, slug } from "../src/util.js"; +import { clamp01, slug, stripTrailingSlashes } from "../src/util.js"; test("clamp01 regression (E5): NaN and non-numeric input fail to 0, never propagate", () => { // Math.max(0, Math.min(1, NaN)) is NaN — one NaN signal poisoned every score it touched. @@ -43,3 +43,13 @@ test("slug regression (E5): non-ASCII names get distinct slugs instead of all co assert.equal(slug("🔥🔥"), a, "deterministic"); for (const s of [arabic, chinese, hindi, a]) assert.ok(!/[\\/:*?"<>|\s]/.test(s), s); }); + +test("stripTrailingSlashes removes only trailing slashes, in linear time", () => { + assert.equal(stripTrailingSlashes("a/b///"), "a/b"); + assert.equal(stripTrailingSlashes("///"), ""); + assert.equal(stripTrailingSlashes("a//b"), "a//b"); + assert.equal(stripTrailingSlashes(null), ""); + const t0 = performance.now(); + assert.equal(stripTrailingSlashes(`${"/".repeat(300000)}x`).length, 300001); + assert.ok(performance.now() - t0 < 1000); +}); From 9ee97d270dfb750677c40d3ee3f54b9c404030bc Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 21:31:27 +0000 Subject: [PATCH 2/4] feat(docs): generate the Mintlify changelog page from CHANGELOG.md The docs site's changelog page was hand-written and stopped at one July entry while thirty releases shipped. It is now generated: - src/changelog_page.js parses CHANGELOG.md and renders every release, plus [Unreleased], as a Mintlify entry: each change's headline (the bold lead, else the first sentence), filter tags per section, and a link to the release's full notes on GitHub. Text is made MDX-safe outside code spans, and relative links point at GitHub. - forge docs render splices it between JSX-comment markers (MDX rejects HTML comments). It is a strict target, so forge docs check fails when the page is stale. - scripts/bump.mjs regenerates the page in the release commit, so a release never leaves it behind (the bump workflow commits with git add -A). All 19 Mintlify pages compile under MDX 3. CHANGELOG corrections that the page surfaced: 1.1.2 called the project's own second-machine re-run an independent replication (now a dated correction), and 1.0.0 had lost the \r\n / \n escapes inside two code spans. GUIDE, ARCHITECTURE, the Mintlify README/CLI page and CLAUDE.md describe the new surface. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- ARCHITECTURE.md | 14 +- CHANGELOG.md | 66 +- CLAUDE.md | 6 +- docs/GUIDE.md | 10 +- mintlify/README.md | 5 + mintlify/changelog/overview.mdx | 1186 +++++++++++++++++++++++++++++-- mintlify/cli/core.mdx | 5 +- scripts/bump.mjs | 5 + src/changelog_page.js | 251 +++++++ src/docs_render.js | 68 +- test/bump.test.js | 24 + test/changelog_page.test.js | 148 ++++ test/docs_render.test.js | 35 + 13 files changed, 1729 insertions(+), 94 deletions(-) create mode 100644 src/changelog_page.js create mode 100644 test/changelog_page.test.js diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 3c55a1ac..2201642d 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -433,8 +433,10 @@ as the `collide_check` MCP tool. **Machine-owned doc surfaces (`src/docs_render.js`, `forge docs render`).** The auto-maintenance layer that keeps tables and diagrams in sync with the code registries. -Four marker-managed blocks (commands table in README, groups and MCP-tools tables in GUIDE, -repo-map diagram in ARCHITECTURE) are regenerated from `COMMANDS`/`GROUPS`/`TOOLS`; six +Five marker-managed blocks (commands table in README, groups and MCP-tools tables in GUIDE, +repo-map diagram in ARCHITECTURE, and the Mintlify changelog page, rendered from +`CHANGELOG.md` by `src/changelog_page.js` between MDX-safe JSX-comment markers) are +regenerated from `COMMANDS`/`GROUPS`/`TOOLS`/`CHANGELOG.md`; six "N MCP tools" count phrases are auto-corrected; and every mermaid block across all `.md` and `.mdx` files receives the branded `%%{init` theme. Registry-derived blocks are CI-gated errors when stale; tree-derived output is advisory. @@ -675,8 +677,8 @@ from the tree it describes. ```mermaid %%{init: {'theme':'base','themeVariables':{'primaryColor':'#201a15','primaryTextColor':'#f2ede7','primaryBorderColor':'#372c22','lineColor':'#f26430','secondaryColor':'#272019','tertiaryColor':'#171310','edgeLabelBackground':'#201a15','clusterBkg':'#171310','clusterBorder':'#4a3b2e','fontFamily':'ui-sans-serif, system-ui, sans-serif','fontSize':'14px'},'flowchart':{'curve':'basis','padding':10,'nodeSpacing':36,'rankSpacing':44}}}%% flowchart LR - test["test
133 files"] - src["src
119 files"] + test["test
134 files"] + src["src
120 files"] landing["landing
61 files"] research["research
37 files"] bench["bench
6 files"] @@ -684,13 +686,13 @@ flowchart LR scripts["scripts
3 files"] docs["docs
1 file"] examples["examples
1 file"] - test -- 281 --> src + test -- 284 --> src bench -- 12 --> src examples -- 4 --> src + scripts -- 3 --> src test -- 3 --> global test -- 3 --> scripts test -- 2 --> bench - scripts --> src src --> global ``` diff --git a/CHANGELOG.md b/CHANGELOG.md index 507dc9a4..6cdce1c7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -74,6 +74,12 @@ evidence behind it. Scripts that read the JSON output may need to adapt: ### Security +- **The claim-registry table escapes backslashes, and four patterns no longer backtrack + quadratically on long runs (CodeQL).** `scripts/claims-status.mjs` escaped `|` in table + cells but not `\`, so a trailing backslash could undo an escape (`js/incomplete-sanitization`, + high). The model-catalog tokenizer's trailing `.0` collapse and its URL trim (`js/polynomial-redos`, + high, pre-existing), the workspace-glob trim and the semantic guard's edge-punctuation trim now + use linear scans with identical results (checked against the old regexes). - **The dashboard checks Host on every route, and writes need the page's session token and this exact origin (F13).** A foreign Host (DNS rebinding) gets 403 on reads too; another localhost port cannot write. Native clients that POSTed without a token must now send the @@ -125,7 +131,13 @@ evidence behind it. Scripts that read the JSON output may need to adapt: ### Added -- `bench/universal-router/reproduce.sh` rebuilds the shipped router prior from pinned, +- **The docs site's changelog page is generated from `CHANGELOG.md`.** It had one hand-written + entry from July while thirty releases shipped. `forge docs render` now writes every release, + plus `[Unreleased]`, as a Mintlify `` entry with each change's headline, filter tags + and a link to its full notes (`src/changelog_page.js`, between MDX-safe JSX-comment markers). + `forge docs check` fails when the page is stale, and `scripts/bump.mjs` regenerates it in the + release commit. +- **`bench/universal-router/reproduce.sh` rebuilds the shipped router prior** from pinned, sha256-checked public inputs (`sources.json`). The refit reproduces `data/router_prior.json` exactly: all 176 fitted values, with only `fittedAt` different (Node v22.22.2, about 7 minutes on 4 vCPUs). `holdout_eval.mjs` is a new seeded 150/350 held-out experiment. It is not a @@ -144,20 +156,33 @@ evidence behind it. Scripts that read the JSON output may need to adapt: ### Documentation -- A machine-readable claim/status registry (`docs/status/claims.json`, 47 claims assessed +- **The Mintlify reference pages describe the current behavior** of `forge verify` (per-package + coverage, pre/post binding, verifier events), `forge stack` (`available` runners), + `forge context` (what `COMPLETE` means, `--block`), `forge reuse` and `forge ledger` + (lossless keys, serve-time revalidation, one vote per event, archive reasons, conflicts, + `--fix --dry-run`), `forge dash` (Host check, session token) and the universal router, which + the site did not document at all. The landing page lists all ten native targets (OpenClaw was + missing). +- **Two historical `CHANGELOG.md` entries are corrected.** 1.1.2 called the project's own + second-machine re-run an independent replication (now marked as a dated correction), and 1.0.0 + had lost the `\r\n` / `\n` escapes inside two code spans. +- **A machine-readable claim/status registry** (`docs/status/claims.json`, 47 claims assessed against `d2abfa6`), with a generated table in `docs/status/README.md`. `node scripts/claims-status.mjs --check`, now part of the CI quality gate, fails when the registry is invalid, the table is stale, or a `docs/cognitive-substrate/` copy has drifted from its `research/` source. -- `docs/INTEGRATIONS.md`: for every supported tool, config emission, MCP registration, - automatic hooks and enforcement are listed separately, each marked tested, declared or not - supported. It also records which registry models have provider ids, with a date. -- Universal router docs. The run-4 held-out headline is labeled repository-reported. The - shipped prior's refit is documented as reproduced exactly in this repository. The new - held-out replay is documented, as are the modeling limits: cascade cost under-predicted by - 5–22%, optimistic targets, and budgets that bound only expected cost. The dataset pin is - corrected to `SWE-bench/SWE-bench_Verified@78f471b`. -- Research corrections (2026-09-26) in the synthesis, the preprint and the white paper: +- **`docs/INTEGRATIONS.md` lists what each tool really gets.** For every supported tool, + config emission, MCP registration, automatic hooks and enforcement are listed separately, + each marked tested, declared or not supported. It also records which registry models have + provider ids, with a date. +- **Universal router docs separate what is measured from what is only reported.** The run-4 + held-out headline is labeled repository-reported. The shipped prior's refit is documented as + reproduced exactly in this repository. The new held-out replay is documented, as are the + modeling limits: cascade cost under-predicted by 5–22%, optimistic targets, and budgets that + bound only expected cost. The dataset pin is corrected to + `SWE-bench/SWE-bench_Verified@78f471b`. +- **The research papers carry dated corrections (2026-09-26).** In the synthesis, the preprint + and the white paper: - Theorem D: joint attainability, with a counterexample. - The equality condition of the silent-miss bound. - "A caught miss is not a completed task." @@ -167,13 +192,13 @@ evidence behind it. Scripts that read the JSON output may need to adapt: `research/recompute_corrections.py --theorem-checks` asserts the Theorem D checks without data, and the recomputation also prints macro F1. -- Evidence grades are split into bibliographic verification, claim support, study design, +- **Evidence grades are split** into bibliographic verification, claim support, study design, independent replication and transfer scope. METR's slowdown result is scoped to its 16-developer, early-2025 study, with a link to the February 2026 update. -- The Qur'anic lens labels the Arabic source text, the translation, tafsir and the author's +- **The Qur'anic lens labels its layers separately:** the Arabic source text, the translation, tafsir and the author's design analogy separately, and states what the lens does and does not establish. No Arabic text or translation was changed. -- The research PDFs are marked as historical, pre-correction editions and recorded in +- **The research PDFs are marked as historical, pre-correction editions** and recorded in `research/HISTORICAL_EDITIONS.md` (git blob, sha256, pinned commit, figure map, render recipe). They were not re-rendered: the Qur'anic text in a fresh render could not be verified, and the refutation paper needs a TeX toolchain. @@ -473,8 +498,11 @@ evidence behind it. Scripts that read the JSON output may need to adapt: ### Added -- **The universal router's benchmark is independently replicated.** harness-bench run 4 was - re-run from the pinned public data (SWE-bench Verified `78f471b`, SWE-bench/experiments +- **The universal router's benchmark was re-run by the project on a second machine.** + _(Corrected 2026-09-26: this entry first said "independently replicated". A re-run by the + project itself is not an independent replication, and the held-out harness lives outside this + repository; see [docs/UNIVERSAL_ROUTING.md](docs/UNIVERSAL_ROUTING.md).)_ harness-bench run 4 + was re-run from the pinned public data (SWE-bench Verified `78f471b`, SWE-bench/experiments `40f164d`) on a second machine. Results: - **Split:** the same 150/350 split. - **Held-out test:** 217 of 218 metrics identical, with only wall-clock fit time differing. @@ -634,10 +662,8 @@ evidence behind it. Scripts that read the JSON output may need to adapt: ### Fixed -- **A claim minted before the CRLF fold is migrated, not deleted.** Folding ` -` into - ` -` changes a claim's content address, so a claim written by an earlier version on a +- **A claim minted before the CRLF fold is migrated, not deleted.** Folding `\r\n` into + `\n` changes a claim's content address, so a claim written by an earlier version on a Windows checkout carried the pre-fold address in its filename and failed its own address check on load — `loadClaims` returned nothing for it, and `forge ledger verify` reported it as an id mismatch. The read path now accepts the pre-fold address as well, so the diff --git a/CLAUDE.md b/CLAUDE.md index bb8f20b5..e5025f84 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -3,7 +3,7 @@ ## Stack - Node.js >=20, pure ESM (`"type": "module"`), zero runtime dependencies. -- Linter/formatter: Biome 2.5.5 (dev dependency). +- Linter/formatter: Biome 2.5.13 (dev dependency; `npx biome migrate --write` after an upgrade). - Types: TypeScript via JSDoc annotations — no `.ts` files, checked by `tsc`. ## Commands @@ -24,4 +24,8 @@ - Run `npm test && npx biome check && npm run typecheck && node src/cli.js docs check` before committing — the docs check fails CI when commands/env vars/MCP tools/CHANGELOG drift from the code, so update docs IN THE SAME CHANGE, not later. +- After editing `CHANGELOG.md` (or commands/MCP tools), run `node src/cli.js docs render`: + it regenerates the machine-owned blocks, including the Mintlify changelog page + (`mintlify/changelog/overview.mdx`), which the docs check fails on when stale. Never edit + between `forge:render` markers by hand. - Version lives in `package.json` — `scripts/bump.mjs` keeps all manifests in sync. diff --git a/docs/GUIDE.md b/docs/GUIDE.md index 8b298ff1..a11c1d56 100644 --- a/docs/GUIDE.md +++ b/docs/GUIDE.md @@ -581,7 +581,12 @@ generated from the same registries the check reads, into marker-managed blocks every diagram in every tracked markdown file re-themes (deliberate bad examples opted out with `docs-check-ignore` are left alone); - the repo map in `ARCHITECTURE.md` — drawn from the live import graph, so it cannot - drift from the tree it describes. + drift from the tree it describes; +- the docs site's changelog page (`mintlify/changelog/overview.mdx`) — from `CHANGELOG.md`: + every release, and the `[Unreleased]` work, as one Mintlify `` entry listing each + change's headline with a link to its full notes. An MDX page cannot hold HTML comments, + so its markers are JSX comments (`{/* forge:render:changelog:begin … */}`). A release + (`scripts/bump.mjs`) regenerates it in the release commit. ```console $ forge docs render @@ -594,7 +599,8 @@ $ forge docs render --check reconciler in CI: a stale registry-derived block is an **error** whose message is the fix (`run forge docs render`), while tree-derived output (the repo map, diagram theme) is a warning — moving a file never fails an unrelated PR, but a new command with a -stale table always does. +stale table always does. The changelog page is an error too: edit `CHANGELOG.md`, then +run `forge docs render` in the same change. ### `forge docs sync` — which prose did this diff make stale? diff --git a/mintlify/README.md b/mintlify/README.md index bc2a8cc2..e5410f55 100644 --- a/mintlify/README.md +++ b/mintlify/README.md @@ -19,6 +19,7 @@ mintlify/ # verification gates, model routing cli/ # overview + one page per command GROUP guides/ # zero-config onboarding, team memory, radar deps + changelog/ # overview.mdx — generated from CHANGELOG.md by `forge docs render` ``` ## Preview locally @@ -54,6 +55,10 @@ handles deploys. The only CI added is an advisory broken-link check file paths relative to this folder, without the `.mdx` extension. - Keep content grounded in the real repo docs — Forge has a "no mock data / metrics must be real" ethos. Do not add invented features or benchmark numbers. +- **Do not edit `changelog/overview.mdx` between its markers.** The entries are generated + from `CHANGELOG.md` (`src/changelog_page.js`): write the release notes there, then run + `node src/cli.js docs render`. `forge docs check` fails CI when the page is stale, and a + release (`scripts/bump.mjs`) regenerates it in the release commit. ## Localization diff --git a/mintlify/changelog/overview.mdx b/mintlify/changelog/overview.mdx index 3d0469c1..201cda4f 100644 --- a/mintlify/changelog/overview.mdx +++ b/mintlify/changelog/overview.mdx @@ -1,57 +1,1141 @@ --- title: "Changelog: what's new in Forge" -description: "Weekly release notes for Forge — new features, updates, and fixes for the CLI, verification gates, and cross-session memory layer." +description: "Every Forge release, newest first: the headline of each change, with a link to its full notes. Generated from CHANGELOG.md." +rss: true --- - - -## Updates - -**`forge verify --deep` now reports a four-state result.** The multi-lens -consensus run no longer collapses to a binary pass/fail. You'll see one of -`PASS`, `FAIL`, `INCOMPLETE`, or `NOT_CONFIGURED`, so a stalled lens or an -unconfigured repo can't quietly read as green. Only `PASS` counts as verified, -and `PASS` now also requires the underlying `forge verify` tests status to be -`PASS` — a green deep run can never outrun a red base run. - -See [Quality commands](/cli/quality) and -[Verification gates](/concepts/verification-gates). - -**Completion gate requires test evidence for code changes.** For any session -that touched source code, the Stop-path completion gate no longer accepts a -`forge handoff` snapshot on its own. You'll need either a test file in the -session's diff or a fresh passing `forge verify` against the current changes. -Doc-only, config-only, and other non-code sessions are unaffected. - -When the gate blocks, the repair checklist now leads with `forge verify` so -the evidence gets recorded before the handoff snapshot. See -[Cross-session memory](/concepts/cross-session-memory). - -**Clearer positioning for hooks and guardrails.** Forge's automatic hooks are -Claude Code-specific, and — like every other advisory signal — they're -off-by-default until you opt in with `FORGE_ENFORCE=1`. The guardrails are -defence in depth, not a security sandbox. Nothing about the runtime changed; -the docs and CLI copy just match what the tool actually does. See -[CLI overview](/cli/overview) and the -[Pre-action gate](/concepts/pre-action-gate). - -## Fixes - -**Secret redactor now preserves structured tool output.** Built-in tools like -Bash, Read, and Grep return structured objects, and the previous redactor -stringified those responses — which meant Claude Code silently discarded the -rewrite and the unredacted output stayed visible. The redactor now walks the -response recursively: strings are scrubbed in place, arrays and objects keep -their keys and shape, and non-string leaves pass through untouched. Plain-string -responses redact byte-identically to before, and a response with nothing to -redact emits no rewrite at all. - -**`FORGE_GUARD_STRICT=1` exit status is no longer swallowed.** The redactor's -shell wrapper used to force a clean exit, hiding the strict-mode failure code -from Claude Code so the `DEGRADED` signal never surfaced. The child's exit -status is now propagated end-to-end, so strict mode's stderr feedback actually -reaches you. Note that this hook runs after the tool has already executed — -strict mode makes redactor degradation loud, but blocking a call before it runs -is the job of the [Pre-action gate](/concepts/pre-action-gate). +Every release, newest first. Each entry lists the headline of every change it shipped; **Full +notes** opens that release in +[CHANGELOG.md](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md), which has the +details, migration notes and the reasoning behind each change. **Unreleased** is what is merged +on the default branch but not yet published to npm. Filter by change type with the tags. + +This page is generated from `CHANGELOG.md` by `forge docs render`, and `forge docs check` fails +CI when it falls behind, so it cannot drift from the release notes again. + +{/* forge:render:changelog:begin (generated by `forge docs render` — do not edit) */} + + +**Changed** + +- **`forge verify` covers the whole repo, or says what it did not cover.** +- **A test runner that is only a devDependency is inventory, not an obligation (F09).** +- **`forge verify` refuses to bind a verdict to code that changed while the tests ran (F10).** +- **The code-state fingerprint is a canonical manifest bound to HEAD (F01).** +- **Exact reuse keys are lossless except whitespace (F04).** +- **`forge context` reports delivery, not availability (F02, F03).** +- **`forge route universal` makes an unreachable objective explicit (F12).** +- **`forge route universal` says when a recommended model cannot be called (A07).** +- **`forge route outcome` validates what it records (A04, A01).** +- **`forge imagine --run` is described as what it is: an isolated checkout of HEAD, not a security sandbox (A05)** +- **Cost model: the 90.2/85.6/74.3% scenarios built on the refuted 0.62 routing factor are withdrawn.** +- **README leads with the three jobs** + +**Security** + +- **The claim-registry table escapes backslashes, and four patterns no longer backtrack quadratically on long runs (CodeQL).** +- **The dashboard checks Host on every route, and writes need the page's session token and this exact origin (F13).** + +**Fixed** + +- **One evidence event, one vote (F06).** +- **A rewritten lesson no longer inherits the old wording's trust (F07).** +- **Artifacts are revalidated where they are served (F05).** +- **Archived is not refuted (F15).** +- **Similar-but-opposite rules are never merged (F16).** +- **Sparse router cost fits no longer invent a $1 intercept (F11).** +- **The reuse benchmark measures the tiers it names (F14).** +- **`forge verify` no longer runs a project's suite with the parent test runner's `NODE_TEST_CONTEXT`** +- The format-on-edit guard no longer rewrites a Biome project with a global prettier. +- Evidence records with a non-finite day or an oversized ref, and malformed `.forge/models.json` entries, are refused at the boundary (reported, never trusted). +- The dashboard reports an unreadable store as unreadable (`meta.errors`), not as empty. +- `forge substrate` says when the impact graph was truncated by the atlas file cap (`capped`, `skippedFiles`), instead of presenting a partial graph as the whole repo (A06). +- `forge verify` no longer counts its own outputs under `.forge/` as changed files. +- `forge ledger verify --fix --dry-run` previews the address migration without writing. + +**Added** + +- **The docs site's changelog page is generated from `CHANGELOG.md`.** +- **`bench/universal-router/reproduce.sh` rebuilds the shipped router prior** +- `src/semantic_guard.js` (behaviour-bearing token comparison) and `src/schema.js` (narrow runtime validation), both zero-dependency. +- Seeded property tests for the trust invariants (`test/trust_properties.test.js`) and a regression test for each reproduced review finding. +- CI runs the two Python prototype suites (49 + 23 tests), the Theorem-D sanity checks and the recomputation of every corrected research number from the shipped replication package. +- Command handlers for memory, verification and routing moved to `src/cli/` (a pure move; `src/cli.js` keeps dispatch and help). + +**Documentation** + +- **The Mintlify reference pages describe the current behavior** +- **Two historical `CHANGELOG.md` entries are corrected.** +- **A machine-readable claim/status registry** +- **`docs/INTEGRATIONS.md` lists what each tool really gets.** +- **Universal router docs separate what is measured from what is only reported.** +- **The research papers carry dated corrections (2026-09-26).** +- **Evidence grades are split** +- **The Qur'anic lens labels its layers separately** +- **The research PDFs are marked as historical, pre-correction editions** +- `forge verify`, `forge stack`, `forge ledger`, `forge reuse`, `forge dash` and `forge route universal` sections of the guide describe the new coverage, binding, archive, reuse, dashboard and advice-only behavior. + +[Full notes for Unreleased →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#unreleased) + + + + + +**Fixed** + +- **deja no longer records host notifications as solved work.** + +[Full notes for v1.4.3 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#143---2026-09-24) + + + + + +**Changed** + +- **`forge uicheck` now exits 1 in three cases that used to exit 0.** + +**Fixed** + +- **`forge uicheck contrast` exits 1 when a pair fails WCAG AA.** +- **`forge uicheck design` no longer passes token-based Tailwind by seeing nothing.** + +[Full notes for v1.4.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#142---2026-09-24) + + + + + +**Fixed** + +- **`protect-paths` matches secret paths per path token, so read-only commands stop being blocked.** +- **`protect-paths` closes the reads it missed.** +- **`protect-paths` fails closed and no longer needs bash.** +- **`forge doctor` no longer asks a plugin user to register every guard twice.** + +[Full notes for v1.4.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#141---2026-09-24) + + + + + +**Fixed** + +- **`forge init` / `forge sync` no longer replace a hand-written `AGENTS.md`, and the Stop-hook auto-sync no longer reverts human edits.** +- **Rules an older version moved to `AGENTS.md.forge-bak` are no longer forgotten.** +- **Size checks cover the whole `AGENTS.md`.** +- **Notes added under a generated `CLAUDE.md` header survive later syncs.** + +**Added** + +- **`forge init --tools ` chooses the agent tools a repo emits config for.** + +**Changed** + +- **`forge init` emits config only for the tools a repo uses by default** +- **`forge integrations add` writes to, and takes ownership in, only the recorded tools' MCP config** +- **`forge tools ` adds the tool to a recorded set that lacks it** + +[Full notes for v1.4.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#140---2026-09-24) + + + + + +**Fixed** + +- **`forge impact`, `forge atlas`, `forge scope` and `forge rank` now follow tsconfig/jsconfig path aliases.** + +[Full notes for v1.3.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#132---2026-09-24) + + + + + +**Fixed** + +- **The Stop gate no longer demands a unit test for a UI-only change.** +- **A passing e2e run counts as test evidence** +- **The Stop gate no longer blames a session for other agents' edits.** +- **`forge substrate`, `forge lean` and `forge anchor` measure only this session's changes** + +[Full notes for v1.3.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#131---2026-09-24) + + + + + +**Added** + +- record which claims were served, and compact by it +- learn retention from the ledger's own history + +[Full notes for v1.3.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#130---2026-09-23) + + + + + +**Added** + +- **`forge ledger compact [--dry-run]`: ledger retention learned from the ledger's own history.** +- **A local usage log** + +**Changed** + +- **The Stop hook's ledger pruning no longer uses a fixed 2 × 45-day window.** +- **Model tiers resolve to the newest live model instead of pinned ids.** +- **`forge models`** +- **`forge route` shows the resolved model id** +- **A catalog held in memory expires when the response says it does.** +- **`forge route gateway` / `forge config gateway`** +- **The custom-gateway remap picks the newest model of each family** +- **The session-log cost estimate no longer bills unknown models at $3/$15.** + +**Security** + +- **A catalog can no longer write anything but a model id into the generated gateway config.** + +[Full notes for v1.2.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#120---2026-09-23) + + + + + +**Added** + +- **The universal router's benchmark was re-run by the project on a second machine.** + +**Fixed** + +- **`fit_prior.mjs` writes its default output on Windows.** + +[Full notes for v1.1.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#112---2026-09-22) + + + + + +**Fixed** + +- **The session learner works on macOS.** +- **Lesson consolidation reads the folder the session learner writes to on Windows.** +- **Sonnet 5 is priced at $2/$10 per million tokens again.** +- **The release job no longer fails while npm catches up.** + +[Full notes for v1.1.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#111---2026-09-22) + + + + + +**Added** + +- **Universal router** + +**Fixed** + +- **Binary files no longer trip the commit gate's secret scan.** +- **A handoff snapshot is read back whole at session start.** +- **The verifyToken example's completeness score matches the code again.** + +**Changed** + +- **The everyday blast-radius checks walk sibling and forward relations, tagged.** +- **`bin/learn-consolidate.sh` no longer lets a model prune memory.** +- **`learn-consolidate.sh --llm` works on macOS.** +- **The gate docs no longer claim that repeated gates multiply their catch rates.** + +[Full notes for v1.1.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#110---2026-09-22) + + + + + +**Added** + +- **`forge ledger verify --fix` re-addresses pre-CRLF-fold claims.** +- **TypeSafe System One (Jev) as the fast proposer.** + +**Fixed** + +- **A claim minted before the CRLF fold is migrated, not deleted.** +- **The impact benchmark's labels are ground truth again, and the numbers they feed are re-measured.** +- **`llm.escalateTo` is no longer advisory-and-inert — a real failure now consumes it.** +- **A CRLF checkout no longer forks a claim id.** +- **`caller_fanout` is no longer dead for callers that only have a path.** +- **`forge impact` actually resolves imports.** +- **The impact graph no longer reads comments and strings as code, and a call belongs to its function.** +- **Building the graph is linear again, and the file cap counts source files.** +- **Blast radius is no longer reverse-only — the refutation's sibling and forward relations are ported.** +- **The impact-quality numbers are re-measured, and they are not the README's.** +- **The in-repo Python prototype is the repaired v2, not the refuted v1.** +- **`forge atlas query` shows the definition you asked for.** +- **Fan-out and churn stop lying.** +- **Non-finite weights are zero, not certainty.** +- **`recommend()` no longer sends a non-finite score to the most expensive tier.** +- **`extractJson` reads the first balanced JSON object, not everything between the first brace and the last.** +- **The gateway model map parses versions instead of matching loose digits.** +- **`classifyIntent` reports the winning intent's confidence, not a losing neighbor's.** +- **`knowledge_router` keeps the first-person signal it routes on.** +- **A null `noul` from Jev is "no answer", not a confident zero, and a choice matches the offered option case-insensitively.** +- **The preflight scanners no longer read addresses, code fences and prose as code.** +- **With the LLM layer on, the assumption gate no longer asks just because a task names something the repo lacks.** +- **A torn ledger line no longer swallows the next record.** +- **Claim canonicalization normalizes keys before sorting them.** +- **An MCP tool that throws now answers with a JSON-RPC error.** +- **`forge ledger sync` no longer erases teammates' evidence from the shared ref.** +- **Restoring a superseded fact leaves it live.** +- **Eq. 3 retrieval ranks by relevance again.** +- **Déjà vu is gated on relevance, not on the total score.** +- **Future-dated evidence no longer counts at full weight for years.** +- **A learned lesson stops flapping out of the injection set the next day.** +- **Dormancy latches, and pruning is wired.** +- **The reuse cache's "exact" tier means the same task again.** +- **The LSH prefilter stopped dropping three of every four adapt candidates.** +- **Goal anchoring measures the right thing, per checkpoint.** +- **The doom-loop signature sees the whole failure.** +- **`LEDGER.md` stopped conflicting in the conflict-free store.** +- **The per-prompt hook stopped re-reading the whole ledger, three times.** +- **A recorded UI interaction verdict is dated today.** +- **A refuted fact stops being broadcast to every tool.** +- **Only a real test run counts as one.** +- **CI is green again on Linux.** +- **The test suite is hermetic.** +- **`verify --deep` no longer claims coverage it never had.** +- **The pre-edit risk advisory can fire.** +- **`forge radar` rings reflect the risk they find.** +- **`forge cost` counts what a session actually cost.** +- **`forge rank` hazard counts incidents, not sessions.** +- **Context assembly stops a source at its 3rd optional item, and only that source.** +- **Non-ASCII names no longer collide, and NaN no longer propagates.** +- **The learning loop runs in real installs again.** +- **Completion-gate evidence can no longer be produced by the agent alone.** +- **The cost governor now governs.** +- **`forge harden` writes a deny list Claude Code actually reads.** +- **The settings allowlist no longer auto-approves three dangerous commands.** +- **Lockfile commits are no longer refused as leaking a secret.** +- **Claude Code hooks no longer fail on Windows with `spawn bash ENOENT`.** +- **`protect-paths` no longer dies (exit 1, fail-open) on machines without `jq`.** + +**Security** + +- **The secret filter can no longer be made to hang.** +- **Credentials in URLs, STS keys and short or slash-bearing values are now caught and masked whole.** +- **The guards parse their payload with a real JSON parser, and fail closed.** +- **Session hook logs no longer store raw secrets, and `init` keeps them out of git.** +- **The commit gate's secret scan now fails closed.** +- **Only evidence forge actually resolved can lift a claim into the trusted band.** +- **Serving a cached artifact no longer confirms it.** +- **The MCP ledger write tools act as the agent and only propose.** + +**Documentation** + +- **The formal synthesis's Theorem D is restated as a bound, and its definitions are fixed.** +- **The refutation paper's statistics are tightened without changing the refutation.** +- **The whitepaper marks its refuted prototype claims in place.** +- **The README and docs stop calling the impact graph conservative and stop presenting 62% as a saving.** +- **`research/recompute_corrections.py` re-derives every corrected number.** +- `CLAUDE.md`: Biome 2.5.2 → 2.5.5 (matching the pin), "600+ tests" → "1000+", and the lint command `npx biome check` → `npm run check` — the documented command fails outright, since the npx package is `@biomejs/biome`, not `biome`. +- **OpenClaw is a first-class emit target — the compiler's tenth tool.** + +**Changed** + +- **`forge impact` stays focused by default; the wide walk is `--all-relations`.** +- **`forge route calibrate` stops calling itself "outcome-calibrated routing".** +- **The routing rubric stops counting a task's length twice and stops matching on one shared word.** +- **Model routing reconciles the proposer's band with the deterministic band, not a point score.** +- **A proposer vote moves the tier only when the proposer is confident.** +- **The proposer can no longer raise the tier.** +- **The assumption gate compares the proposer's verdict with the rubric's instead of clipping one scale onto the other.** +- **MCP targets address their server bucket by dotted key path.** + +[Full notes for v1.0.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#100---2026-09-22) + + + + + +**Fixed** + +- **Plugin load: dropped the duplicate `hooks` declaration from the manifest.** + +[Full notes for v0.32.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0321---2026-08-22) + + + + + +**Changed** + +- **Landing: source-owned technical instrument.** + +[Full notes for v0.32.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0320---2026-08-14) + + + + + +**Changed** + +- **impact: hazard-aware blast radius.** + +**Fixed** + +- **docs: refresh post-v0.30.0 staleness.** + +[Full notes for v0.31.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0310---2026-08-07) + + + + + +**Added** + +- **`forge collide` — the parallel-session conflict radar.** + +**Fixed** + +- **The rank hazard join now matches production claims.** +- **`beliefDiff` tombstone edge cases.** +- **`forge rank` determinism and hardening.** +- **Temporal CLI guards.** + +[Full notes for v0.30.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0300---2026-08-07) + + + + + +**Added** + +- **`forge docs render` — the docs that can write themselves, do.** + +[Full notes for v0.29.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0290---2026-08-07) + + + + + +**Added** + +- **`forge rank` — load-bearing code, measured.** +- **Time-travel for team memory.** +- **Merkle state root.** + +**Changed** + +- **`impact()` dequeues in O(1).** +- **Lesson glob compilation is memoized.** + +[Full notes for v0.28.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0280---2026-08-07) + + + + + +**Fixed** + +- **CI is green again.** + +[Full notes for v0.27.4 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0274---2026-08-04) + + + + + +**Changed** + +- **The Mintlify docs site is now English-only.** + +[Full notes for v0.27.3 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0273---2026-07-21) + + + + + +**Changed** + +- **Synced the docs with the code and added a Mintlify drift guard.** + +[Full notes for v0.27.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0272---2026-07-21) + + + + + +**Security** + +- **The `forge dash` write routes are now guarded against CSRF and DNS-rebinding.** + +[Full notes for v0.27.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0271---2026-07-21) + + + + + +**Changed** + +- **The PCM ledger is now the DEFAULT memory store — the legacy-store migration is finished (M2).** + +**Fixed** + +- **Closed the ledger-only read/write gaps the default flip exposed (M2).** + +[Full notes for v0.27.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0270---2026-07-20) + + + + + +**Fixed** + +- **`reconcileFacts` no longer risks wiping memory under `FORGE_LEDGER_ONLY` (M2).** + +[Full notes for v0.26.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0262---2026-07-20) + + + + + +**Changed** + +- **Split the `run()` god-function in `src/cli.js` into a dispatch table (H3).** + +[Full notes for v0.26.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0261---2026-07-20) + + + + + +**Added** + +- **`problem-solver` skill** + +**Fixed** + +- **Broke a static ESM import cycle in the memory layer.** +- **Importing the package as a library no longer executes the CLI.** +- **Hermetic tests.** +- **The unimplemented-command stub now exits non-zero** +- Replaced literal NUL bytes with `\0` escapes in the `reuse.js`/`diagnose.js` content-hash separators (byte-identical runtime output; the source is now clean text). + +**Changed** + +- **The release pipeline is now closed-loop.** +- **Consolidated duplicated helpers into `src/util.js`.** +- Removed the dead `plugin` entry from `package.json` `files`, the vestigial `.gitlab-ci.yml`, and the dangling `@AGENTS.md` import in `CLAUDE.md`; fixed the off-palette Codex plugin `brandColor`. +- **Broke a layering cycle in `repo_config.js`.** +- `ledger_store.js`'s `git cat-file -e` ref resolver now passes `--` before the ref (defense in depth against a ref that begins with `-`). +- **Refreshed `ARCHITECTURE.md`** + +[Full notes for v0.26.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0260---2026-07-20) + + + + + +**Changed** + +- **Landing + status pages: token-driven fluid type scale and spacing scale.** + +[Full notes for v0.25.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0250---2026-07-20) + + + + + +**Added** + +- **`forge docs impact` — a reusable documentation-impact graph.** + +[Full notes for v0.24.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0240---2026-07-20) + + + + + +**Fixed** + +- **Windows path portability (Git Bash CI).** + +[Full notes for v0.23.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0232---2026-07-20) + + + + + +**Fixed** + +- **`forge imagine --dry-run` per-file attribution on macOS.** + +[Full notes for v0.23.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0231---2026-07-19) + + + + + +**Security** + +- **PostToolUse redaction now honors the structured-output contract (CR-01).** +- **Strict mode is honest and actually propagates (CR-02).** +- **Guard blocks shell writes to protected paths (HI-06).** +- **Hardened git secret-reader detection (HI-07).** +- **Broader hook tool coverage (HI-08).** + +**Fixed** + +- **`forge verify` runs every detected suite (HI-01).** +- **Verification is bound to the final code state (HI-02).** +- **Completion gate detects edits to pre-dirty files (HI-03).** +- **A changed test file is no longer proof (HI-04).** +- **Honest spawn and metrics classification (ME-01, ME-02).** +- **Ownership-safe settings uninstall (HI-05).** +- **Init aborts on profile-persistence failure (HI-09).** +- **Safer installer transactions (HI-10).** +- **Verification surfaces monorepo suites (ME-03).** +- **Evidence resolution strength gates confidence (ME-05).** +- **Secret-bearing ledger metadata is refused (ME-06).** +- **Trusted quarantine identity (ME-07).** +- **Per-target MCP ownership and atomic add/remove (ME-08, ME-10).** +- **Divergent registry-name MCP entries preserved (ME-09).** +- **Validated MCP names and safe serialization (ME-11).** +- **Non-object JSON is treated as corrupt (ME-12).** +- **Crash-safe config writes (ME-13).** +- **Complete tool list (ME-14).** +- **Functional doctor health probes (ME-15, ME-16, ME-17, ME-18).** +- **Explicit PARTIAL sync status (ME-19).** +- **Fail-safe legacy rules (ME-20).** +- **Exact gitattributes rule detection (ME-21).** + +**Changed** + +- **Disclose the global settings merge before it happens (ME-22).** +- **Hooks install in exec form (ME-23).** +- **Honest positioning (ME-24).** + +[Full notes for v0.23.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0230---2026-07-19) + + + + + +**Changed** + +- **Only real profiles (RA-14).** +- **Honest wording (RA-22, RA-23, RA-24).** + +**Security** + +- **Git read-command bypasses closed (RA-05).** +- **Secret redactor no longer fails silently (RA-06).** + +**Fixed** + +- **`forge verify --deep` can no longer pass when nothing ran (RA-01).** +- **`forge verify` executes the detected runner (RA-08).** +- **CLI verify output distinguishes all four states (RA-09).** +- **Ledger log lines are hash-verified at read time (RA-02).** +- **Stale ambient atlas is no longer authoritative (RA-07).** +- **Completion gate enforces real obligations (RA-10).** +- **Doctor honesty (RA-19, RA-20).** +- **Settings failures propagate (RA-04).** +- **Hook paths are shell-quoted (RA-12).** +- **Invalid profile aborts before side effects (RA-13).** +- **Reversible settings uninstall (RA-11, RA-17, RA-18).** +- **MCP emitters are no longer destructive (RA-03).** +- **MCP configs respect ownership (RA-21).** +- **Stop-hook auto-sync detects body drift (RA-16).** +- **Repo config consolidated; malformed JSON fails loudly (RA-15).** + +[Full notes for v0.22.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0221---2026-07-18) + + + + + +**Fixed (audit remediation)** + +- **Onboarding & state safety.** +- **Security defaults.** +- **Managed-file integrity.** +- **Third-party MCP is opt-in.** +- **Core correctness.** +- **Effective-date pricing.** +- **Release integrity.** +- **Honest claims & wording.** + +**Added / changed (product)** + +- **Policy profiles & repo config.** +- **Change-obligation guidance.** +- **Subsystem health.** +- **Help grouping.** +- **UI rule.** + +[Full notes for v0.22.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0220---2026-07-17) + + + + + +**Fixed** + +- **`forge deja` could never fire.** +- **`forge harden` crashed in a linked worktree/submodule.** +- **`forge ledger sync` could silently skip a push to a new/pruned remote.** +- **`forge verify --deep` evidence base could diverge.** +- **Cortex halt metrics were always "pass".** +- **`forge radar` usage misattribution.** +- **`forge precommit` diff-header parse.** + +[Full notes for v0.21.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0211---2026-07-17) + + + + + +**Added** + +- **Per-command help + word forms.** +- **`forge doctor --fix` — one-command auto-repair.** +- **Zero-config settings wiring + first-run hint.** +- **`forge tools` — primary-tool config + auto-gitignore for secondary-tool artifacts.** +- **`forge dash` first-run clarity.** +- **Web session-start install hook.** + +[Full notes for v0.21.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0210---2026-07-17) + + + + + +**Added** + +- **`forge know` — the A7 knowledge-router.** +- **Commit-level gate rung (`forge precommit`).** +- **Pin/downgrade via `update --to `.** +- **Multi-lens verification consensus.** +- **`forge radar` — dependency-currency rings (I4 verified currency).** +- **Cross-machine memory sync.** +- **Anti-repetition memory (`forge deja`).** +- **Dashboard v2 (`forge dash`).** +- **Static HTML report (`forge report`).** + +[Full notes for v0.20.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0200---2026-07-17) + + + + + +**Added** + +- **Color-aware CLI output.** + +**Fixed** + +- **Research crosswalk reconciled with the code.** + +[Full notes for v0.19.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0190---2026-07-17) + + + + + +**Changed** + +- **Single design-token source.** +- **Landing page redesign.** +- **Accessibility + design-system hygiene.** + +**Added** + +- **Social + icon metadata on both public pages.** +- **Brand-aligned status line.** +- **Mermaid theme-value guard.** + +**Fixed** + +- **Status-page metrics were silently stale.** +- **Generated status page no longer ships stale/leaky.** +- **Landing → status link** + +[Full notes for v0.18.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0180---2026-07-16) + + + + + +**Added** + +- **OpenAI + Gemini provider detection** + +[Full notes for v0.17.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0170---2026-07-15) + + + + + +**Added** + +- **Playwright interaction loop** + +[Full notes for v0.16.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0160---2026-07-15) + + + + + +**Added** + +- measured-promotion gate + outcome-calibrated routing + +[Full notes for v0.15.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0150---2026-07-15) + + + + + +**Added** + +- **Measured-promotion gate + outcome-calibrated routing (`forge route calibrate`)** +- **Legacy-store retirement (`FORGE_LEDGER_ONLY`)** + +[Full notes for v0.14.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0140---2026-07-15) + + + + + +**Added** + +- Custom-gateway model remap (`src/gateway_model_map.js`). + +[Full notes for v0.13.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0130---2026-07-15) + + + + + +**Fixed** + +- security: drop two inert `curl`-pipe deny rules from the settings template. + +[Full notes for v0.12.4 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0124---2026-07-11) + + + + + +**Fixed** + +- bump.mjs keeps ROADMAP's "Now" marker in sync + +[Full notes for v0.12.3 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0123---2026-07-11) + + + + + +**Fixed** + +- allowlist bibliography citation-key false positives in gitleaks +- don't let an empty Unreleased section blank the status page changelog + +[Full notes for v0.12.2 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0122---2026-07-11) + + + + + +**Fixed** + +- don't let an empty Unreleased section blank the status page changelog + +[Full notes for v0.12.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0121---2026-07-11) + + + + + +**Changed** + +- **Goal-drift classification is graded and identifier-aware (`src/anchor.js`).** +- **Specification completeness is a logistic estimator (`src/preflight.js`).** + +**Added** + +- **`forge docs check` now guards intra-repo links and roadmap freshness** + +**Fixed** + +- **Dead and fabricated docs** + +[Full notes for v0.12.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0120---2026-07-11) + + + + + +**Added** + +- **`forge stack`** +- **Six more atlas languages** +- **`forge update`** +- **Self-dogfood** +- **Auto-release on merge** +- **`forge docs check` now guards diagrams, model prices, and benchmark numbers** + +**Changed** + +- **CLI output is quiet by default** +- **Unified public design system** +- **Restyled the terminal statusline** +- **Plain-language docs pass** + +**Fixed** + +- **Broken diagrams** +- **Status page "Latest changes" list** + +[Full notes for v0.11.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0110---2026-07-11) + + + + + +**Added** + +- **The completion gate** +- **`forge handoff`** +- **`forge decide`** +- **`forge docs sync`** +- **Session baseline + rehydration** +- **Intent protocol cards** +- **Recorded assumptions** +- **Config artifacts in the atlas** +- **End-to-end skills + agent** + +**Fixed** + +- **`cortex.sh` hook entry resolution in symlink installs** +- **Twelve defects found by a two-angle adversarial review of the new layer, all with regression tests** + +[Full notes for v0.10.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#0100---2026-07-10) + + + + + +**Added** + +- **Gateway environments work end to end** +- **`forge docs check`** +- **Docs are in the impact graph** +- **Persistent goal** +- **AGENTS.md auto-repair** +- **Entropy secret detection** +- **`src/math.js`** + +**Changed** + +- **Routing scores by exemplar similarity, not keyword lists** +- **Lesson matching is graded** +- **Substrate minimality warnings derive from computed signals** +- **`forge scan` detects obfuscated payloads** +- **`providerStatus` probes `/health` on any custom base URL** + +[Full notes for v0.9.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#090---2026-07-10) + + + + + +**Added** + +- **MCP write tools** + +**Changed** + +- Simplified CLI surface and improved dashboard UX empty states. + +**Fixed** + +- Stale documentation across command references. + +[Full notes for v0.8.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#081---2026-07-08) + + + + + +**Added** + +- **Forge work system** +- **Zero-config provider auto-detection** +- **Hosted LiteLLM gateway support** + +**Fixed** + +- TypeScript errors and Biome 2.5.2 lint warnings across source and tests. + +[Full notes for v0.8.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#080---2026-07-08) + + + + + +**Added** + +- **Optional embeddings tier** +- **`forge uicheck visual `** + +**Changed** + +- **Ledger read-path flip (P2).** +- **Professional redesign of the public site, gated by forge's own UI system.** + +[Full notes for v0.7.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#070---2026-07-08) + + + + + +**Changed** + +- Docs consolidation pass: deduplicated cross-doc prose into single canonical homes (the substrate README now points at the GUIDE's command reference, output table, and honest-limits list instead of repeating them), added orientation diagrams (ARCHITECTURE four-layer compiler + … + +[Full notes for v0.6.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#060---2026-07-07) + + + + + +**Added** + +- Security & OSS hardening: CodeQL, gitleaks secret-scan (blocking; verified clean on the full history), and OSSF Scorecard workflows; refreshed repo topics; SECURITY.md now states the supported line (0.5.x) and documents the ledger's forgery-resistance properties (content-hash … +- **UI fingerprints resolve CSS `var()` indirection** +- **One-click release automation.** +- **Benchmark harness (`npm run bench`) + measured results doc.** +- **Loop closure (P5 of the substrate-v2 plan): doom-loop diagnosis, imagination, CUSUM drift, checkpoint cadence.** +- **Context assembly + completeness gate (P4 of the substrate-v2 plan).** +- **Generated-UI quality gate (P6 of the substrate-v2 plan).** +- **Local dashboard (P7 of the substrate-v2 plan).** +- **Measured cost report (P8 of the substrate-v2 plan).** +- **Proof-carrying reuse cache (P3 of the substrate-v2 plan).** +- **Team memory (P2 of the substrate-v2 plan).** +- **Proof-Carrying Memory ledger (P1 of the substrate-v2 plan).** +- **Uniform `--json`.** +- **`forge doctor` sees more silent misconfiguration.** +- **Evaluation harness (`src/eval.js`).** +- **Opt-in LLM adjudication for the substrate (`FORGE_LLM=1`)** +- **Bidirectional verified reconcile (default on when `FORGE_LLM=1`; `llm.bidirectional` in `source/substrate.json` to disable).** +- **Explicit memory `val` term** + +**Fixed** + +- **PCM ledger hardened after an 8-angle adversarial review of the P1 merge.** +- **The Cortex capture/learn loop now works in the dotfile install too.** +- **`forge verify` can't hang.** +- **Secret-refusal no longer guts auth-related work.** +- **One malformed file no longer takes down memory.** +- **`recordMistake` reports `refused` (not `created`) when a save is rejected** +- **Atlas emits `inherits` edges** +- **Atlas is incremental + staleness-aware.** +- **Performance** +- **`substrate` no longer recomputes preflight twice** + +**Documentation** + +- **Substrate v2 plan: the whitepaper, completed (`docs/plans/substrate-v2/`).** +- **Visual flow diagrams in the entry-point docs.** + +**Changed** + +- **Model tiers carry a currency + a verified date.** +- **One shared call-site extractor (`src/extract.js`).** +- **Opt-in enforcing gate (`FORGE_ENFORCE=1`).** +- **M5 anti-over-engineering is now measured, not guessed (`forge lean`).** +- **Doom-loop breaker (self-correction).** +- **Consequence simulation — failing-tests class (Eq 4).** +- **`forge sync` now adopts an existing project `CLAUDE.md` instead of skipping it.** +- **Unified the model-call path** + +[Full notes for v0.5.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#050---2026-07-07) + + + + + +**Added** + +- **Forge Cognitive Substrate** +- **M4 goal-anchoring (`forge anchor`)** +- **Atlas v2 graph** +- **`docs/GUIDE.md`** +- **Repo automation** + +**Changed** + +- **Publish to public npm.** +- **Substrate auto-runs in Claude Code** +- **Docs overhaul** + +**Fixed** + +- **Security (research prototype)** +- **Smaller npm package** +- **Perf** + +[Full notes for v0.4.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#040---2026-07-06) + + + + + +**Changed** + +- **Publish to GitHub Packages** + +[Full notes for v0.3.1 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#031---2026-07-05) + + + + + +**Added** + +- **Forge Preflight** + +[Full notes for v0.3.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#030---2026-07-05) + + + + + +**Added** + +- **Forge Cortex** +- Ambient hooks (fail-safe, never block): capture signals during a session, distill at `Stop`, inject learned lessons at `SessionStart`, and a `PreToolUse` advisory before a risky edit. +- Local error predictor (heuristic + a tiny logistic model) gated by an AUC-PR kill-switch — it only ships if it measurably beats the heuristic; otherwise it falls back or disables. +- Cross-tool: lessons inlined into `AGENTS.md` + a zero-dependency MCP server (`forge cortex-mcp`, registered in `source/mcp.json`). +- Optional LLM lesson distiller (`ENABLE_CORTEX_DISTILL=1`) — replaces the deterministic template with a real distilled lesson via `claude -p`. +- `forge doctor` reports Cortex lesson state; `forge catalog` lists Cortex. + +[Full notes for v0.2.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#020---2026-07-05) + + + + + +**Added** + +- Cross-tool config emitter (`forge sync`) — one source → each tool's native format; three install channels (Claude plugin + marketplace, installer, npm); `forge doctor`; code-graph (`atlas`); `lean` discipline; guard/skill/crew layers. +- Verification layer: `forge verify` (tests + hallucinated-symbol catch + provenance), doom-loop breaker guard, bias-safe `independent-reviewer` agent. +- Security gate: `forge scan` (skill-gate), `secret-redact` guard, structured `permissionDecision` in `protect-paths`, `forge harden` (gitleaks + sandbox). +- Cross-tool MCP emit; portable memory (`forge brain` / `forge remember`); design-taste menu (`forge taste`); `forge spec` spec-lock + OpenSpec wiring; MCP ~6-server hygiene check; coverage + type-checking (`tsc --checkJs`); 2026 production-standard rules; OWASP-LLM / NIST SSDF / … + +[Full notes for v0.1.0 →](https://github.com/CodeWithJuber/forgekit/blob/HEAD/CHANGELOG.md#010---2026-07-05) +{/* forge:render:changelog:end */} diff --git a/mintlify/cli/core.mdx b/mintlify/cli/core.mdx index 17ac93c0..598405f0 100644 --- a/mintlify/cli/core.mdx +++ b/mintlify/cli/core.mdx @@ -114,7 +114,7 @@ Docs ↔ code drift. ```bash forge docs check # registry reconcile — commands, env vars, MCP tools, CHANGELOG -forge docs render # regenerate machine-owned tables, counts, and diagrams from the registries +forge docs render # regenerate machine-owned tables, counts, diagrams, and the changelog page forge docs sync # diff-driven stale-docs sweep ``` @@ -122,7 +122,8 @@ forge docs sync # diff-driven stale-docs sweep `docs check` fails CI when commands, env vars, MCP tools, or the CHANGELOG drift from the code — and when a generated block is stale, the error names the fix: `forge docs render` regenerates the command tables, MCP tool table, count phrases, shared mermaid - theme, and the repo map from the code registries. `docs sync` sweeps the diff and + theme, and the repo map from the code registries, and this site's changelog page from + `CHANGELOG.md`. `docs sync` sweeps the diff and reports UPDATED / STALE / VERIFIED-UNAFFECTED. diff --git a/scripts/bump.mjs b/scripts/bump.mjs index d95d1952..0ffe5ad5 100644 --- a/scripts/bump.mjs +++ b/scripts/bump.mjs @@ -32,6 +32,8 @@ import { execFileSync } from "node:child_process"; import fs from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; +import { CHANGELOG_PAGE } from "../src/changelog_page.js"; +import { renderFile } from "../src/docs_render.js"; // --------------------------------------------------------------------------- // Pure version math @@ -356,6 +358,9 @@ export function applyBump(root, currentVersion, newVersion, date) { const changelog = readIfExists(path.join(root, clRel)); if (changelog !== null) { write(clRel, rotateChangelog(changelog, newVersion, currentVersion, date)); + // The docs site's changelog page is generated from CHANGELOG.md (docs check fails when + // it is stale), so the release commit regenerates it: [Unreleased] became this version. + if (renderFile(root, CHANGELOG_PAGE, { write: true }).changed) changed.push(CHANGELOG_PAGE); } return changed; diff --git a/src/changelog_page.js b/src/changelog_page.js new file mode 100644 index 00000000..379af4b3 --- /dev/null +++ b/src/changelog_page.js @@ -0,0 +1,251 @@ +// The docs site's changelog page, generated from CHANGELOG.md so the two cannot drift. The +// page used to be written by hand and stopped at one July entry while thirty releases +// shipped. Now every release — and the [Unreleased] work on the default branch — becomes one +// Mintlify entry listing the headline of each change (the bold lead sentence the +// CHANGELOG convention puts first; the first sentence when there is none), with a link to +// that release's full notes. `forge docs render` splices it into the page between MDX-safe +// markers, `forge docs check` fails when the page is stale, and scripts/bump.mjs +// regenerates it in the release commit. Pure text → text, node stdlib only. +import { readFileSync } from "node:fs"; +import { join } from "node:path"; + +/** The page the entries are spliced into (relative to the repo root). */ +export const CHANGELOG_PAGE = "mintlify/changelog/overview.mdx"; + +const MONTHS = [ + "January", + "February", + "March", + "April", + "May", + "June", + "July", + "August", + "September", + "October", + "November", + "December", +]; + +/** "2026-09-24" → "September 24, 2026" (no locale dependency); anything else unchanged. */ +export function longDate(iso) { + const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(String(iso ?? "")); + if (!m || Number(m[2]) < 1 || Number(m[2]) > 12) return String(iso ?? ""); + return `${MONTHS[Number(m[2]) - 1]} ${Number(m[3])}, ${m[1]}`; +} + +/** GitHub's heading anchor: lowercase, punctuation dropped, each space a hyphen. */ +export function githubSlug(heading) { + return String(heading) + .trim() + .toLowerCase() + .replace(/[^\p{L}\p{N}\s_-]/gu, "") + .replace(/\s/g, "-"); +} + +/** + * Parse Keep-a-Changelog text into releases, newest first. Each section keeps the first + * paragraph of each top-level bullet (joined onto one line); nested lists, later paragraphs, + * prose between bullets and fenced code are not headlines and are skipped. Repeated section + * headings within one release merge. + * @param {string} text + * @returns {{version: string, date: string|null, heading: string, + * sections: {label: string, bullets: string[]}[]}[]} + */ +export function parseChangelog(text) { + /** @type {ReturnType} */ + const releases = []; + /** @type {ReturnType[number] | null} */ + let release = null; + /** @type {{label: string, bullets: string[]} | null} */ + let section = null; + /** @type {string[] | null} */ + let bullet = null; + let fenced = false; + const flush = () => { + if (bullet && section) section.bullets.push(bullet.join(" ")); + bullet = null; + }; + const sectionFor = (label) => { + if (!release) return null; + let s = release.sections.find((x) => x.label === label); + if (!s) { + s = { label, bullets: [] }; + release.sections.push(s); + } + return s; + }; + for (const line of String(text ?? "") + .replace(/\r\n/g, "\n") + .split("\n")) { + if (/^\s*(```|~~~)/.test(line)) { + flush(); + fenced = !fenced; + continue; + } + if (fenced) continue; + const h2 = /^## \[([^\]]+)\](?:\s+-\s+(\S+))?/.exec(line); + if (h2) { + flush(); + release = { + version: h2[1], + date: h2[2] ?? null, + heading: line.slice(3).trim(), + sections: [], + }; + releases.push(release); + section = null; + continue; + } + if (!release) continue; + if (/^\[[^\]]+\]:\s/.test(line)) { + // The compare-link references at the bottom end the last release. + flush(); + release = null; + continue; + } + const h3 = /^###\s+(.+?)\s*$/.exec(line); + if (h3) { + flush(); + section = sectionFor(h3[1]); + continue; + } + if (line.startsWith("- ")) { + flush(); + section ??= sectionFor("Changes"); + bullet = [line.slice(2).trim()]; + continue; + } + if (bullet && /^\s+\S/.test(line) && !/^\s+(?:[-*+]|\d+\.)\s/.test(line)) { + bullet.push(line.trim()); + continue; + } + flush(); + } + flush(); + return releases; +} + +/** Cut `s` to at most `max` characters at a word boundary, never inside a code span. */ +function clip(s, max) { + if (s.length <= max) return s; + let cut = s.lastIndexOf(" ", max); + if (cut < max / 2) cut = max; + let out = s.slice(0, cut); + if ((out.match(/`/g) ?? []).length % 2) out = out.slice(0, out.lastIndexOf("`")).trimEnd(); + return `${out} …`; +} + +/** + * The headline of one bullet: its bold lead, or else its first sentence (outside code + * spans, not at "e.g."/"i.e."), clipped to `max` characters. + * @param {string} bullet the bullet's first paragraph, on one line + * @param {{max?: number}} [opts] + */ +export function headline(bullet, { max = 280 } = {}) { + const t = String(bullet).replace(/\s+/g, " ").trim(); + const bold = /^\*\*(.+?)\*\*/.exec(t); + // A lead that introduces a list ("…by default:") reads as a sentence without its colon. + if (bold) return `**${bold[1].trim().replace(/\s*[:;,]$/, "")}**`; + let code = false; + for (let i = 0; i < t.length; i++) { + const c = t[i]; + if (c === "`") code = !code; + else if ( + !code && + (c === "." || c === "!" || c === "?") && + (i + 1 === t.length || t[i + 1] === " ") && + !/\b(?:e\.g|i\.e|etc|vs|approx)$/i.test(t.slice(0, i)) + ) + return clip(t.slice(0, i + 1), max); + } + return clip(t.replace(/\s*[:;,]$/, ""), max); +} + +const ESCAPES = { "<": "<", ">": ">", "{": "{", "}": "}" }; + +/** + * Markdown → MDX-safe markdown for one line: outside code spans, `<` `>` `{` `}` become + * character references (MDX would read them as JSX or expressions); autolinks become links; + * relative links point at the file on GitHub, so the site never links into its own tree. + * @param {string} s + * @param {string} repo repository web URL, e.g. https://github.com/owner/name + */ +export function mdxInline(s, repo) { + return String(s) + .split(/(`[^`]*`)/) + .map((part, i) => { + if (i % 2) return part; // a code span is literal in MDX + return part + .replace(/<(https?:\/\/[^>\s]+)>/g, "[$1]($1)") + .replace(/\]\(([^)\s]+)\)/g, (_, href) => { + if (/^(?:https?:|mailto:)/i.test(href)) return `](${href})`; + if (href.startsWith("#")) return `](${repo}/blob/HEAD/CHANGELOG.md${href})`; + return `](${repo}/blob/HEAD/${href.replace(/^\.?\//, "")})`; + }) + .replace(/[<>{}]/g, (c) => ESCAPES[/** @type {"<"|">"|"{"|"}"} */ (c)]); + }) + .join(""); +} + +/** The filter tag of a section heading: "Fixed (audit remediation)" → "Fixed". */ +const tagOf = (label) => { + const w = /^[A-Za-z]+/.exec(label)?.[0] ?? "Changes"; + return w[0].toUpperCase() + w.slice(1).toLowerCase(); +}; + +/** + * Every release as a Mintlify entry, newest first. + * @param {string} changelog the CHANGELOG.md text + * @param {{repo: string}} opts repository web URL (for the "full notes" links) + * @returns {string} + */ +export function renderChangelogUpdates(changelog, { repo }) { + const out = []; + for (const r of parseChangelog(changelog)) { + const sections = r.sections.filter((s) => s.bullets.length); + if (!sections.length) continue; + const unreleased = /^unreleased$/i.test(r.version); + const label = unreleased ? "Unreleased" : `v${r.version}`; + const description = unreleased + ? "Merged on the default branch, not yet released" + : longDate(r.date); + const tags = [...new Set(sections.map((s) => tagOf(s.label)))]; + out.push( + ``, + "", + ); + for (const s of sections) { + out.push(`**${mdxInline(s.label, repo)}**`, ""); + for (const b of s.bullets) out.push(`- ${mdxInline(headline(b), repo)}`); + out.push(""); + } + out.push( + `[Full notes for ${label} →](${repo}/blob/HEAD/CHANGELOG.md#${githubSlug(r.heading)})`, + "", + "", + "", + ); + } + return out.join("\n").trimEnd(); +} + +/** The repository's web URL from package.json `repository` (git+https / .git stripped). */ +export function repoUrl(root) { + try { + const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf8")); + const url = typeof pkg.repository === "string" ? pkg.repository : pkg.repository?.url; + return String(url ?? "") + .replace(/^git\+/, "") + .replace(/\.git$/, "") + .replace(/^git@github\.com:/, "https://github.com/"); + } catch { + return ""; + } +} + +/** The generated block for `root`'s CHANGELOG.md (what `forge docs render` splices in). */ +export function renderChangelogPage(root) { + const changelog = readFileSync(join(root, "CHANGELOG.md"), "utf8"); + return renderChangelogUpdates(changelog, { repo: repoUrl(root) }); +} diff --git a/src/docs_render.js b/src/docs_render.js index 5bcec781..09741642 100644 --- a/src/docs_render.js +++ b/src/docs_render.js @@ -2,8 +2,9 @@ // DETECT drift between the registries and the prose; every fix was still a human // hand-editing tables in five files. This module closes the loop: the derivable parts // of the docs (command tables, the MCP tool table, the "N MCP tools" count phrases, the -// mermaid theme, the repo map) are RENDERED from the same registries the check reads — -// COMMANDS/GROUPS, mcp_tools.TOOLS, brand.json, the live import graph — into +// mermaid theme, the repo map, the docs site's changelog page) are RENDERED from the same +// sources the check reads — COMMANDS/GROUPS, mcp_tools.TOOLS, brand.json, the live import +// graph, CHANGELOG.md — into // marker-managed blocks, exactly the pattern bench/bench.mjs already uses for // reports/benchmarks.md. Prose stays human; tables and diagrams become machine-owned. // `forge docs render` regenerates; `--check` (and the docs-check reconciler) fails CI @@ -15,14 +16,20 @@ import { readFileSync, writeFileSync } from "node:fs"; import { join } from "node:path"; import { BRAND } from "./brand.js"; +import { CHANGELOG_PAGE, renderChangelogPage } from "./changelog_page.js"; import { commandSummary, GROUPS } from "./commands.js"; import { TOOLS } from "./mcp_tools.js"; import { directedImportGraph } from "./scope.js"; import { git } from "./util.js"; -const BEGIN = (name) => - ``; -const END = (name) => ``; +// Markdown blocks use HTML comments; MDX (the Mintlify site) rejects those, so an MDX block's +// markers are JSX comments instead. +const BEGIN = (name, mdx = false) => + mdx + ? `{/* forge:render:${name}:begin (generated by \`${BRAND.cli} docs render\` — do not edit) */}` + : ``; +const END = (name, mdx = false) => + mdx ? `{/* forge:render:${name}:end */}` : ``; /** Pad a markdown table so the raw text stays readable (the repo's tables are aligned). */ function mdTable(headers, rows) { @@ -131,12 +138,13 @@ export function renderRepoMap(root, { maxDirs = 9 } = {}) { return `\`\`\`mermaid\n${lines.join("\n")}\n\`\`\``; } -/** Replace one managed block. Returns the new text plus whether markers were found/changed. */ -export function spliceBlock(text, name, body) { - const begin = text.indexOf(BEGIN(name)); - const end = text.indexOf(END(name)); +/** Replace one managed block. Returns the new text plus whether markers were found/changed. + * `mdx` selects the JSX-comment markers an .mdx page needs. */ +export function spliceBlock(text, name, body, { mdx = false } = {}) { + const begin = text.indexOf(BEGIN(name, mdx)); + const end = text.indexOf(END(name, mdx)); if (begin === -1 || end === -1 || end < begin) return { text, found: false, changed: false }; - const next = `${text.slice(0, begin)}${BEGIN(name)}\n${body}\n${text.slice(end)}`; + const next = `${text.slice(0, begin)}${BEGIN(name, mdx)}\n${body}\n${text.slice(end)}`; return { text: next, found: true, changed: next !== text }; } @@ -187,6 +195,15 @@ const BLOCK_TARGETS = [ render: (root) => renderRepoMap(root), strict: false, }, + { + // Every release as a Mintlify entry — the page used to be hand-written and fell + // thirty releases behind. Strict: a CHANGELOG change must re-render it. + file: CHANGELOG_PAGE, + name: "changelog", + render: (root) => renderChangelogPage(root), + strict: true, + mdx: true, + }, ]; // Files whose "N MCP tools" phrases are auto-corrected (the six the count lives in). @@ -233,14 +250,14 @@ export function renderDocs(root = BRAND.root, { write = false } = {}) { for (const t of BLOCK_TARGETS) { const doc = load(t.file); if (!doc) continue; - if (doc.text.indexOf(BEGIN(t.name)) === -1) { + if (doc.text.indexOf(BEGIN(t.name, t.mdx)) === -1) { // No markers → nothing to manage here (fixture roots, forks that opted out). // Don't render (the repo map walks the tree) and don't flag — the registry // reconcilers still cover the content the old hand-written way. missing.push({ file: t.file, name: t.name }); continue; } - const r = spliceBlock(doc.text, t.name, t.render(root)); + const r = spliceBlock(doc.text, t.name, t.render(root), { mdx: t.mdx }); if (!r.found) { missing.push({ file: t.file, name: t.name }); continue; @@ -281,3 +298,30 @@ export function renderDocs(root = BRAND.root, { write = false } = {}) { // strict surfaces fail the check. return { ok: files.every((f) => !f.strict), files, missing }; } + +/** + * Re-render the managed blocks of ONE file — how the release script regenerates the + * changelog page in the release commit without touching any other doc surface. + * @param {string} root + * @param {string} file repo-relative path + * @param {{write?: boolean}} [opts] + * @returns {{found: boolean, changed: boolean}} + */ +export function renderFile(root, file, { write = false } = {}) { + let text; + try { + text = readFileSync(join(root, file), "utf8"); + } catch { + return { found: false, changed: false }; + } + let next = text; + let found = false; + for (const t of BLOCK_TARGETS) { + if (t.file !== file || next.indexOf(BEGIN(t.name, t.mdx)) === -1) continue; + const r = spliceBlock(next, t.name, t.render(root), { mdx: t.mdx }); + found ||= r.found; + next = r.text; + } + if (write && next !== text) writeFileSync(join(root, file), next); + return { found, changed: next !== text }; +} diff --git a/test/bump.test.js b/test/bump.test.js index cd0433bf..65e7300e 100644 --- a/test/bump.test.js +++ b/test/bump.test.js @@ -20,6 +20,7 @@ import { setUnreleasedBody, synthesizeChangelog, } from "../scripts/bump.mjs"; +import { BRAND } from "../src/brand.js"; // --------------------------------------------------------------------------- // version math @@ -382,3 +383,26 @@ test("rotateChangelog refuses an empty [Unreleased] — a release must describe /\[Unreleased\] is empty/, ); }); + +test("applyBump regenerates the docs site's changelog page in the release commit", () => { + const dir = makeFixture(); + try { + fs.writeFileSync( + path.join(dir, "package.json"), + '{\n "name": "fixture",\n "version": "0.4.0",\n "repository": "https://github.com/o/r"\n}\n', + ); + const rel = "mintlify/changelog/overview.mdx"; + fs.mkdirSync(path.join(dir, "mintlify", "changelog"), { recursive: true }); + fs.writeFileSync( + path.join(dir, rel), + `---\ntitle: x\n---\n\n{/* forge:render:changelog:begin (generated by \`${BRAND.cli} docs render\` — do not edit) */}\n{/* forge:render:changelog:end */}\n`, + ); + const changed = applyBump(dir, "0.4.0", "0.5.0", "2026-07-07"); + assert.ok(changed.includes(rel), changed.join(", ")); + const page = fs.readFileSync(path.join(dir, rel), "utf8"); + assert.match(page, /\` command.** It does {a} and , for example + across two lines. + + A second paragraph that is not the headline. +- A plain bullet without a bold lead. It has a second sentence. + - a nested item that is not a headline + +\`\`\` +- a fenced line that is not a bullet +\`\`\` + +### Fixed (audit remediation) + +- **Lead that introduces a list:** + - item +- See [the guide](docs/GUIDE.md#section) and [the notes](#210---2026-09-30), or . + +### Added + +- **A second Added heading merges into the first.** + +## [2.0.0] - 2026-01-02 + +### Changed + +- **Everything, e.g. the API, changed.** Details. + +[Unreleased]: https://github.com/owner/name/compare/v2.1.0...HEAD +[2.1.0]: https://github.com/owner/name/compare/v2.0.0...v2.1.0 +- a link-reference trailer is not part of any release +`; + +test("parseChangelog: releases, merged sections, first paragraphs only", () => { + const r = parseChangelog(SAMPLE); + assert.deepEqual( + r.map((x) => [x.version, x.date]), + [ + ["Unreleased", null], + ["2.1.0", "2026-09-30"], + ["2.0.0", "2026-01-02"], + ], + ); + assert.deepEqual(r[0].sections, [], "an empty [Unreleased] has no sections"); + const v210 = r[1]; + assert.deepEqual( + v210.sections.map((s) => s.label), + ["Added", "Fixed (audit remediation)"], + ); + const added = v210.sections[0].bullets; + assert.equal(added.length, 3, "two Added headings merge; nested and fenced lines are skipped"); + assert.equal( + added[0], + "**A new `thing ` command.** It does {a} and , for example across two lines.", + ); + assert.ok(!added.join(" ").includes("second paragraph")); + assert.ok(!added.join(" ").includes("fenced line")); + assert.ok(!r[2].sections[0].bullets.join(" ").includes("link-reference trailer")); +}); + +test("headline: bold lead, else first sentence — code spans and abbreviations respected", () => { + assert.equal(headline("**Bold lead.** Rest of it."), "**Bold lead.**"); + assert.equal(headline("**Lead that introduces a list:**"), "**Lead that introduces a list**"); + assert.equal(headline("Plain first. Second."), "Plain first."); + assert.equal(headline("Use `a. b` here. Then more."), "Use `a. b` here."); + assert.equal(headline("It changed, e.g. the API. Then more."), "It changed, e.g. the API."); + const long = `Word ${"`code span` ".repeat(60)}end`; + const cut = headline(long, { max: 100 }); + assert.ok(cut.endsWith(" …") && cut.length <= 102, cut); + assert.equal((cut.match(/`/g) ?? []).length % 2, 0, "never cuts inside a code span"); +}); + +test("mdxInline: escapes JSX/expression characters outside code, rewrites repo links", () => { + assert.equal(mdxInline("a {c} `d {f}`", REPO), "a <b> {c} `d {f}`"); + assert.equal( + mdxInline("[g](docs/GUIDE.md#x) [n](#210) [w](https://w.org) ", REPO), + `[g](${REPO}/blob/HEAD/docs/GUIDE.md#x) [n](${REPO}/blob/HEAD/CHANGELOG.md#210) [w](https://w.org) [https://e.com](https://e.com)`, + ); +}); + +test("dates and anchors match what readers and GitHub see", () => { + assert.equal(longDate("2026-09-24"), "September 24, 2026"); + assert.equal(longDate("not a date"), "not a date"); + assert.equal(githubSlug("[1.4.3] - 2026-09-24"), "143---2026-09-24"); + assert.equal(githubSlug("[Unreleased]"), "unreleased"); +}); + +test("renderChangelogUpdates: one Update per non-empty release, tagged, linked to full notes", () => { + const out = renderChangelogUpdates(SAMPLE, { repo: REPO }); + const updates = out.match(/^$/gm) ?? []; + assert.equal(updates.length, 2, "the empty [Unreleased] is skipped"); + assert.equal( + updates[0], + '', + ); + assert.match(out, /^- \*\*A new `thing ` command\.\*\*$/m); + assert.match(out, /^- A plain bullet without a bold lead\.$/m); + assert.match(out, /^\*\*Fixed \(audit remediation\)\*\*$/m); + assert.ok(out.includes(`(${REPO}/blob/HEAD/CHANGELOG.md#210---2026-09-30)`)); + assert.equal((out.match(/^<\/Update>$/gm) ?? []).length, 2); +}); + +test("the repository's changelog page is current and MDX-safe", () => { + const page = readFileSync(join(BRAND.root, CHANGELOG_PAGE), "utf8"); + const begin = page.indexOf("{/* forge:render:changelog:begin"); + const end = page.indexOf("{/* forge:render:changelog:end */}"); + assert.ok(begin !== -1 && end > begin, "the page carries the MDX render markers"); + const block = page.slice(page.indexOf("\n", begin) + 1, end).trimEnd(); + assert.equal(block, renderChangelogPage(BRAND.root), "run `forge docs render`"); + const released = [ + ...readFileSync(join(BRAND.root, "CHANGELOG.md"), "utf8").matchAll(/^## \[(\d+\.\d+\.\d+)\]/gm), + ]; + const newest = released[0]?.[1]; + assert.ok(block.includes(`label="v${newest}"`), `the newest release (${newest}) is on the page`); + // Outside code spans, the only JSX is the Update element and its tags expression. + for (const line of block.split("\n")) { + const prose = line.replace(/`[^`]*`/g, ""); + if (/^$/.test(line)) continue; + assert.doesNotMatch(prose, /[<{}]/, `unescaped MDX character in: ${line}`); + } +}); diff --git a/test/docs_render.test.js b/test/docs_render.test.js index eaf2928f..985ad076 100644 --- a/test/docs_render.test.js +++ b/test/docs_render.test.js @@ -10,6 +10,7 @@ import { normalizeMermaid, renderCommandsTable, renderDocs, + renderFile, renderGroupsTable, renderMcpToolsTable, renderRepoMap, @@ -125,3 +126,37 @@ test("renderDocs on a root without markers manages nothing and stays ok (fixture "absence is reported as informational", ); }); + +test("an MDX page's block uses JSX-comment markers; renderFile re-renders only that file", () => { + const root = dir(); + mkdirSync(join(root, "mintlify", "changelog"), { recursive: true }); + writeFileSync( + join(root, "CHANGELOG.md"), + "# Changelog\n\n## [Unreleased]\n\n## [1.0.0] - 2026-09-01\n\n### Added\n\n- **First release.**\n", + ); + writeFileSync( + join(root, "package.json"), + JSON.stringify({ repository: { url: "git+https://github.com/o/r.git" } }), + ); + const page = join(root, "mintlify", "changelog", "overview.mdx"); + const begin = `{/* forge:render:changelog:begin (generated by \`${BRAND.cli} docs render\` — do not edit) */}`; + writeFileSync(page, `---\ntitle: x\n---\n\n${begin}\n{/* forge:render:changelog:end */}\n`); + // An HTML-comment marker would break MDX; the block is found by its JSX markers only. + assert.equal(spliceBlock(readFileSync(page, "utf8"), "changelog", "x").found, false); + assert.equal( + spliceBlock(readFileSync(page, "utf8"), "changelog", "x", { mdx: true }).found, + true, + ); + const stale = renderDocs(root, { write: false }); + assert.ok(!stale.ok, "an unrendered changelog page is strict drift"); + assert.ok(stale.files.some((f) => f.file === "mintlify/changelog/overview.mdx" && f.strict)); + assert.deepEqual(renderFile(root, "mintlify/changelog/overview.mdx", { write: true }), { + found: true, + changed: true, + }); + const text = readFileSync(page, "utf8"); + assert.match(text, / Date: Sat, 26 Sep 2026 21:31:36 +0000 Subject: [PATCH 3/4] docs: sync the Mintlify reference, landing page and claim registry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit After the 2026-09-26 review fixes merged (#164), several surfaces still described the old behavior: - Mintlify reference: forge verify (per-package coverage, pre/post binding, verifier events, .forge/forge.config.json), forge stack (available runners), forge context (what COMPLETE means, --budget, --block), forge reuse and forge ledger (lossless keys, serve-time revalidation, one vote per event, archive reasons, conflicts, --fix --dry-run), forge dash (Host check on every route, session token), and the universal router, which the site did not cover at all (objectives, INFEASIBLE, advice-only models, outcome provenance, and the limits of its evidence). - Landing page: all ten native targets (OpenClaw was missing, "Nine native targets"). The grid is now two rows of five on wide screens and two columns below 1180px, checked in headless Chromium at eight widths with no overflow or horizontal scroll. - Claim registry: the claims the review found refuted at d2abfa6 are re-assessed on 7eef611 (the merge), after re-running the review's probes as regression tests (46 pass; 1 skip needs a non-root user): verify binding, context completeness and exact reuse are implemented (with their scope in the notes); evidence independence and outcome provenance are partial; the imagine sandbox and router target claims stay refuted. The impact fixture's F1 is 0.28 after the CLI move. - biome.json migrated to the installed Biome 2.5.13 (schema URL; `recommended` → `preset`); lint results unchanged. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- biome.json | 4 +- docs/status/README.md | 28 ++--- docs/status/claims.json | 113 +++++++++++--------- landing/index.html | 11 +- mintlify/cli/config.mdx | 18 +++- mintlify/cli/memory.mdx | 20 ++++ mintlify/cli/quality.mdx | 22 ++++ mintlify/cli/substrate.mdx | 25 ++++- mintlify/concepts/model-routing.mdx | 42 +++++++- mintlify/concepts/proof-carrying-memory.mdx | 14 ++- mintlify/concepts/verification-gates.mdx | 22 ++++ 11 files changed, 240 insertions(+), 79 deletions(-) diff --git a/biome.json b/biome.json index 10ae15a1..9ed2716f 100644 --- a/biome.json +++ b/biome.json @@ -1,5 +1,5 @@ { - "$schema": "https://biomejs.dev/schemas/2.5.2/schema.json", + "$schema": "https://biomejs.dev/schemas/2.5.13/schema.json", "vcs": { "enabled": true, "clientKind": "git", "useIgnoreFile": true }, "files": { "ignoreUnknown": true, @@ -22,7 +22,7 @@ "linter": { "enabled": true, "rules": { - "recommended": true, + "preset": "recommended", "suspicious": { "noAssignInExpressions": "off" } diff --git a/docs/status/README.md b/docs/status/README.md index 005228c1..23fb96ce 100644 --- a/docs/status/README.md +++ b/docs/status/README.md @@ -34,8 +34,8 @@ The registry exists because the project's own rule — *do not assert without ev place where every assertion's evidence can be looked up. -47 claims — implemented 10 · measured 10 · partial 3 · reported 3 · hypothesis 6 · refuted 15. -Assessed against commit `d2abfa69fb77` (as of 2026-09-26). +47 claims — implemented 13 · measured 10 · partial 5 · reported 3 · hypothesis 6 · refuted 10. +Assessed against commits `d2abfa69fb77`, `7eef61179d16` (as of 2026-09-26). | ID | Status | Claim | Component · version | Evidence | Notes | | --- | --- | --- | --- | --- | --- | @@ -44,7 +44,7 @@ Assessed against commit `d2abfa69fb77` (as of 2026-09-26). | `impact-oracle-repaired-heldout` | **measured** | The repaired oracle's F1 on three held-out repositories is 0.416 against grep's 0.371 at the canonical threshold 0.02. | Python impact oracle (research/python-prototypes/impact_oracle) · v2, repaired (in-tree, 49 tests) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/impact_oracle/README.md`](../../research/python-prototypes/impact_oracle/README.md) | A point estimate (ΔF1 about +0.044). Per-repository counts exist only at threshold 0.10 (pooled ΔF1 +0.0565, 3/3 repositories favour it, one-sided sign test p = 0.125); pytest supplies 71.3% of held-out pairs. Never mix the 0.02 and 0.10 figures. | | `impact-oracle-repaired-beats-grep` | **hypothesis** | The repaired oracle beats a grep baseline on repositories it has not seen. | Python impact oracle (research/python-prototypes/impact_oracle) · v2, repaired | [`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md)
[`research/README.md`](../../research/README.md) | Not established: three held-out repositories, and the choice of which relations to add was made after diagnosing all nine (an architecture-selection channel into the test set). Next study: freeze parser and relation design before acquiring new repositories or a time split; report review-cost metrics beside F1. | | `impact-oracle-in-tree-is-v2` | **implemented** | The in-tree Python impact oracle is the repaired v2, with 49 tests (36 demo-package + 13 repair regressions). | Python impact oracle (research/python-prototypes/impact_oracle) · v2 | [`research/python-prototypes/impact_oracle/tests/test_repair_fixes.py`](../../research/python-prototypes/impact_oracle/tests/test_repair_fixes.py)
[`research/python-prototypes/impact_oracle/tests/test_demo_package.py`](../../research/python-prototypes/impact_oracle/tests/test_demo_package.py) | research/README.md said until 2026-09-26 that the repaired oracle shipped only in the replication archive. ImpactOracle(wm, sibling_enabled=False, forward_enabled=False) reproduces the refuted v1 traversal. | -| `node-impact-fixture-quality` | **measured** | forge impact scores precision 0.17, recall 1.00, F1 0.29 on six hand-labelled cases from this repository. | Node code graph (src/atlas.js, forge impact) · 1.4.3 | [`reports/benchmarks.md`](../../reports/benchmarks.md)
[`bench/impact_cases.mjs`](../../bench/impact_cases.mjs) | Six self-labelled symbols in one repository; the transitive walk is scored against direct-only labels. The 2026-09-26 review measured macro 0.18 / 1.00 / 0.30 on the same cases. A regex-derived graph, not the evaluated Python AST oracle; none of the Python study's numbers apply to it. | +| `node-impact-fixture-quality` | **measured** | forge impact scores precision 0.17, recall 1.00, F1 0.28 on six hand-labelled cases from this repository. | Node code graph (src/atlas.js, forge impact) · master after 1.4.3 (unreleased) | [`reports/benchmarks.md`](../../reports/benchmarks.md)
[`bench/impact_cases.mjs`](../../bench/impact_cases.mjs) | Six self-labelled symbols in one repository; the transitive walk is scored against direct-only labels. Re-measured on the tree merged as 7eef611, after the CLI handler move added src/cli/memory.js as an eleventh claimText dependent (reports/benchmarks.md: mean F1 0.28, was 0.29). The 2026-09-26 review measured macro 0.18 / 1.00 / 0.30 on the same cases at d2abfa6. A regex-derived graph, not the evaluated Python AST oracle; none of the Python study's numbers apply to it. | | `router-gate-demo-saving` | **refuted** | Complexity routing saves 62.1% of cost against always using the premium tier. | Old tiered router (research/python-prototypes/router_gate) · July 2026 prototype, thresholds tuned on 30 tasks | [`research/python-prototypes/router_gate/eval_results.json`](../../research/python-prototypes/router_gate/eval_results.json)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Measured on the 30 tasks its thresholds were tuned on, by repricing measured tokens (a price counterfactual). On 80 held-out tasks its total spend was 20.2% higher. | | `router-gate-heldout-cost` | **measured** | On 64 non-halted held-out tasks the routed pipeline spent $6.3582 against always-premium's $5.2893: 20.21% more. | Old tiered router (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/empirical-refutation/README.md`](../../research/empirical-refutation/README.md) | Escalation-inclusive and ungated. 58 of 64 tasks failed at every tier, which makes the inversion largely mechanical. | | `router-gate-judge-accepted-cost` | **measured** | Per judge-accepted output the routed pipeline cost $1.060 against always-premium's $1.763. | Old tiered router (research/python-prototypes/router_gate) · as evaluated (replication archive) | [`research/empirical-refutation/replication_package.tar.gz`](../../research/empirical-refutation/replication_package.tar.gz)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py)
[`research/python-prototypes/router_gate/README.md`](../../research/python-prototypes/router_gate/README.md) | judge_accepted only: 6/64 vs 3/64 (Clopper-Pearson [0.035, 0.193] and [0.010, 0.131]), so the ratio is unstable. The judge is also the mid-tier executor. tests_passed, human_accepted and deployed_without_revert were not measured. | @@ -56,25 +56,25 @@ Assessed against commit `d2abfa69fb77` (as of 2026-09-26). | `universal-router-inrepo-replay-vs-best-single` | **measured** | On a new in-repo split (seed 20260926; 150 dev / 350 held-out tasks), the router solves 80.0% of held-out tasks at $0.124 per task, against 76.0% at $0.768 for the best single model chosen on dev (claude-opus-4.5). | Universal router (src/router, forge route universal) · dev fit of 150 tasks (holdout_eval.mjs), router code of d2abfa6 and aedddf5 | [`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | +4.0 points [+0.6, +7.4] at -$0.644 per task [-$0.698, -$0.594], mostly from one cheap model: minimax-m2.5 opens 308 of the 350 cascades. It depends on which model wins dev: across six splits the solve-rate interval excludes zero only for this seed. The replay assumes a perfect, free check between attempts; one scaffold, twelve Python repositories, February 2026 costs, and SWE-bench Verified's label-validity limits. | | `universal-router-prior-refit` | **measured** | The shipped prior (data/router_prior.json) refits exactly from pinned public data with this repository's own tooling. | Universal router prior (data/router_prior.json) · fitted 2026-09-22, k=1, scale 4 | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/reproduce.sh`](../../bench/universal-router/reproduce.sh)
[`bench/universal-router/sources.json`](../../bench/universal-router/sources.json)
[`bench/universal-router/compare_priors.mjs`](../../bench/universal-router/compare_priors.mjs)
[`bench/universal-router/fit_prior.mjs`](../../bench/universal-router/fit_prior.mjs) | Reproduced on 2026-09-26 by the project in its own environment (Node v22.22.2, Python 3.11.15, 4-vCPU Intel Xeon @ 2.80GHz): from an empty work directory all 176 fitted values are identical, only provenance.fittedAt differs, with the router code of d2abfa6 and again with aedddf5; the refit takes 413-424 s. The external review's refit stopped at a 180 s limit, which was too short; the 37.6 s / 45.9 s in the 2026-09-22 report were the 150-issue dev fit. Tasks: Hugging Face SWE-bench/SWE-bench_Verified at revision 78f471b (not princeton-nlp/SWE-bench_Verified). reproduce.sh and sources.json were added after the assessed commit. A calculation reproduction, not an independent third-party replication. | | `universal-router-prior-transfer` | **hypothesis** | The shipped prior predicts success and cost on a user's own workload. | Universal router prior (data/router_prior.json) · fitted 2026-09-22 | [`data/router_prior.json`](../../data/router_prior.json)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[openai.com/index/why-we-no-longer-evaluate-swe-bench-verified](https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/) | Fitted on all 500 SWE-bench Verified tasks (so it cannot be evaluated out of sample on them), one scaffold, Python repositories, February 2026 costs. OpenAI reports test flaws in 59.4% of an audited 138-problem hard subset (not of all 500) plus contamination evidence. Next: time-held-out or unseen repositories, another scaffold, another language, real spend. | -| `universal-router-target-guarantee` | **refuted** | The target:p objective guarantees a success rate of at least p. | Universal router (src/router, forge route universal) · 1.4.3 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist. | -| `universal-router-budget-contract` | **partial** | The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists. | Universal router (src/router, forge route universal) · 1.4.3 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`src/router/policy.js`](../../src/router/policy.js) | B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). At the assessed commit an infeasible budget returned the least expensive cascade with no budgetMet:false (review F12); aedddf5 (committed after the assessed commit) reports feasible:false, budgetMet:false, minimumExpectedCost, maxPossibleCost and a reason, and the CLI prints INFEASIBLE. | -| `universal-router-cascade-cost-independence` | **refuted** | A later cascade attempt costs the same in expectation whether or not earlier attempts failed. | Universal router cost model (src/router/policy.js) · 1.4.3 | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | The cascade cost formula assumes E[cost_i \| earlier attempts failed, x] = E[cost_i \| x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes. | -| `route-outcome-provenance` | **refuted** | Outcomes recorded with forge route outcome are verified results. | Universal router outcome log (src/router/index.js) · 1.4.3 | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`src/router/index.js`](../../src/router/index.js) | At the assessed commit the pass/fail was caller-entered, with no provenance or attempt identity, so replaying a log multiplied evidence. aedddf5 (committed after the assessed commit) labels records provenance: self-reported unless --verify-run names a forge verify run whose verdict agrees (then verify-event), makes recording idempotent by attemptId (--attempt), and schema-validates records on write and read. Re-assess before changing the status. | +| `universal-router-target-guarantee` | **refuted** | The target:p objective guarantees a success rate of at least p. | Universal router (src/router, forge route universal) · master after 1.4.3 (unreleased) | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`bench/universal-router/README.md`](../../bench/universal-router/README.md) | Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist. Unchanged at 7eef611. | +| `universal-router-budget-contract` | **partial** | The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists. | Universal router (src/router, forge route universal) · master after 1.4.3 (unreleased) | [`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md)
[`src/router/policy.js`](../../src/router/policy.js) | B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). The explicit half now holds at 7eef611: an infeasible budget returns ok:false, feasible:false, budgetMet:false, minimumExpectedCost and a reason (the CLI prints INFEASIBLE), with the least-bad cascade only as a labeled fallback, and every recommendation reports maxPossibleCost (review F12, regression-tested). | +| `universal-router-cascade-cost-independence` | **refuted** | A later cascade attempt costs the same in expectation whether or not earlier attempts failed. | Universal router cost model (src/router/policy.js) · master after 1.4.3 (unreleased) | [`bench/universal-router/README.md`](../../bench/universal-router/README.md)
[`bench/universal-router/holdout_eval.mjs`](../../bench/universal-router/holdout_eval.mjs)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | The cascade cost formula assumes E[cost_i \| earlier attempts failed, x] = E[cost_i \| x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes. Unchanged at 7eef611. | +| `route-outcome-provenance` | **partial** | Outcomes recorded with forge route outcome are verified results. | Universal router outcome log (src/router/index.js) · master after 1.4.3 (unreleased) | [`src/router/index.js`](../../src/router/index.js)
[`test/router_universal.test.js`](../../test/router_universal.test.js)
[`docs/UNIVERSAL_ROUTING.md`](../../docs/UNIVERSAL_ROUTING.md) | Re-assessed on 7eef611: records are schema-validated on write and read, recording is idempotent by attemptId (--attempt), and a row is provenance verify-event only when --verify-run names a forge verify run whose verdict agrees. Without --verify-run a row is labeled self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable). | | `cost-reduction-90-target` | **hypothesis** | Forgekit reduces coding-agent cost by about 90%. | Cost model (docs/plans/substrate-v2/05-cost-model.md) · plan, corrected 2026-09-26 | [`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`reports/cost-eval.md`](../../reports/cost-eval.md) | The owner's target. The 90.2 / 85.6 / 74.3% scenarios were derived from the refuted 0.62 routing factor and were withdrawn on 2026-09-26; separately estimated stage savings do not multiply into a total. No end-to-end cost has been measured. | | `phase-p0-specs` | **implemented** | P0: specs 00-08, ADR-0005 and ADR-0006 are merged. | Substrate v2 plan · v0.5.0 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/adr/0005-allow-runtime-dependencies.md`](../../docs/adr/0005-allow-runtime-dependencies.md)
[`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md) | | | `phase-p1-ledger-core` | **implemented** | P1: the claim ledger stores content-addressed claims whose confidence moves only with independent oracle evidence. | Ledger (src/ledger.js, src/ledger_store.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger.js`](../../src/ledger.js)
[`test/ledger.test.js`](../../test/ledger.test.js) | The review found evidence-semantics defects at the assessed commit: aliases of one git object counted as independent evidence (F06), and a rewritten lesson inherited the old statement's confidence (F07); aedddf5 (committed after the assessed commit) repairs both with regression tests. See ledger-evidence-independence. | | `phase-p2-team-sync` | **implemented** | P2: ledgers merge across teammates by git union, converging regardless of order. | Ledger sync (src/ledger_sync.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/ledger_sync.js`](../../src/ledger_sync.js)
[`test/ledger_sync.test.js`](../../test/ledger_sync.test.js) | Storage convergence, property-tested; not semantic agreement or factual correctness. The read-path flip (merged legacy + ledger view) and ledger-only writes shipped after P2. | -| `phase-p3-reuse-cache` | **implemented** | P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one. | Reuse cache (src/reuse.js, forge reuse) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js) | Acceptance counterexamples at the assessed commit: exact keys erased operators and case (F04), and changed or deleted artifacts still exact-hit (F05). aedddf5 (committed after the assessed commit) repairs both with regression tests; re-assess before relying on it, and mark this phase partial if the probes still fail. | -| `phase-p4-context-assembly` | **partial** | P4: context assembly never exceeds its token budget and reports a computed missing set. | Context assembly (src/context.js, forge context) · 1.4.3 | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/context.js`](../../src/context.js) | Over budget and pointer-only coverage were reported as complete (F02, F03). aedddf5 (committed after the assessed commit) reports overflow and pending reads; tokens are a chars/3.6 estimate of the rendered block; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. | +| `phase-p3-reuse-cache` | **implemented** | P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one. | Reuse cache (src/reuse.js, forge reuse) · master after 1.4.3 (unreleased) | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: the acceptance counterexamples found at d2abfa6, exact keys that erased operators and case (F04) and changed or deleted artifacts that still exact-hit (F05), are repaired with regression tests and a seeded property test. | +| `phase-p4-context-assembly` | **partial** | P4: context assembly never exceeds its token budget and reports a computed missing set. | Context assembly (src/context.js, forge context) · master after 1.4.3 (unreleased) | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/context.js`](../../src/context.js) | Re-assessed on 7eef611: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, and a pointer is not coverage (F02, F03). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered. | | `phase-p5-loop-closure` | **implemented** | P5: outcomes move the confidence of the claims that informed an action; repeated failures mint a diagnosis; imagine dry-runs the selected tests. | Loop closure (src/diagnose.js, src/imagine.js) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/imagine.js`](../../src/imagine.js)
[`test/imagine.test.js`](../../test/imagine.test.js) | imagine --run is an isolated git checkout of the committed baseline, not a security sandbox, and runs selected files under node --test. | | `phase-p6-ui-quality-gate` | **implemented** | P6: forge uicheck flags a known-template fixture and passes a project-conformant one, with no LLM calls. | UI checks (src/uicheck.js, forge uicheck) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`docs/plans/substrate-v2/07-ui-quality-gate.md`](../../docs/plans/substrate-v2/07-ui-quality-gate.md)
[`test/uicheck.test.js`](../../test/uicheck.test.js) | The checks are advisory measurements (contrast, token conformance, template distance, fingerprint similarity); none measures accessibility or user value, and none is a blocking hook. | | `phase-p7-dashboard` | **implemented** | P7: forge dash renders the ledger, cost meter, cache rate and blast radius offline. | Dashboard (src/dash.js, forge dash) · 1.4.3 | [`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md)
[`src/dash.js`](../../src/dash.js)
[`test/dash.test.js`](../../test/dash.test.js) | The review found read routes answering a foreign Host header at the assessed commit (F13), repaired in aedddf5 (committed after the assessed commit). Cost panels show stage self-estimates, not end-to-end spend. | | `phase-p8-evaluation` | **partial** | P8: a measured, not asserted, end-to-end cost figure per stage is published in reports/. | Cost evaluation (src/cost_report.js, forge cost --stages) · 1.4.3 | [`reports/cost-eval.md`](../../reports/cost-eval.md)
[`docs/plans/substrate-v2/05-cost-model.md`](../../docs/plans/substrate-v2/05-cost-model.md)
[`docs/plans/substrate-v2/00-overview.md`](../../docs/plans/substrate-v2/00-overview.md) | Stage instrumentation and the stage report exist; no paired end-to-end run has been measured and reports/cost-eval.md holds no data. | -| `context-completeness` | **refuted** | forge context COMPLETE means the required knowledge was delivered within the token budget. | Context assembly (src/context.js, forge context) · 1.4.3 | [`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | At the assessed commit a 1-token budget returned ok:true with 39 tokens (F02), and a one-line pointer counted as delivering a definition below line 100 (F03). The contract repaired in aedddf5 (committed after the assessed commit): ok means syntactically delivered, overflow and pending reads are reported, and semantic sufficiency is not measured. | -| `verify-pass-binding` | **refuted** | A forge verify PASS is bound to the exact code state that was tested and covers every declared suite. | Verification (src/verify.js, forge verify) · 1.4.3 | [`src/verify.js`](../../src/verify.js)
[`test/verify.test.js`](../../test/verify.test.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Counterexamples at the assessed commit: the working-tree fingerprint ignored paths and file boundaries (F01), a failing nested workspace still gave a whole-repo PASS (F08), and code changed during the run was signed (F10). aedddf5 (committed after the assessed commit) uses a length-delimited manifest, reports uncovered workspaces as INCOMPLETE and detects mutation; re-run the review's probes before changing this status. | -| `ledger-evidence-independence` | **refuted** | A claim's confidence rises only with independent evidence for that claim. | Ledger (src/ledger_store.js, src/ledger_bridge.js) · 1.4.3 | [`src/ledger_store.js`](../../src/ledger_store.js)
[`src/ledger_bridge.js`](../../src/ledger_bridge.js) | At the assessed commit four prefixes of one commit raised confidence past the 0.8 threshold (F06), and a contradictory rewrite kept its evidence and 0.64 confidence (F07). aedddf5 (committed after the assessed commit) repairs this with regression tests; re-run the review's probes before changing the status. | -| `reuse-exact-identity` | **refuted** | An exact reuse hit serves an artifact minted for the same specification, and only while its bytes still match. | Reuse cache (src/reuse.js) · 1.4.3 | [`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js) | At the assessed commit '>= 18' and '<= 18' shared an exact key (F04), and an edited or deleted artifact still exact-hit (F05). aedddf5 (committed after the assessed commit) repairs this with regression tests; re-run the review's probes before changing the status. | -| `imagine-sandbox` | **refuted** | forge imagine --run executes the selected tests in a sandbox. | Consequence simulation (src/imagine.js, forge imagine) · 1.4.3 | [`src/imagine.js`](../../src/imagine.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | It is an isolated git checkout (a detached-HEAD worktree): it isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. Since aedddf5 (committed after the assessed commit) the CLI says so ("an isolated checkout of HEAD; not a security sandbox"), and a suite that is not node:test gets an explicit unsupported-runner result. | +| `context-completeness` | **implemented** | forge context COMPLETE means every required item was delivered as content within the (estimated) token budget. | Context assembly (src/context.js, forge context) · master after 1.4.3 (unreleased) | [`src/context.js`](../../src/context.js)
[`test/context.test.js`](../../test/context.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/plans/substrate-v2/04-context-assembly.md`](../../docs/plans/substrate-v2/04-context-assembly.md) | Re-assessed on 7eef611: F02 (over budget reported as ok) and F03 (a pointer counted as coverage) are regression tests, and a seeded property test holds the rendered block within budget for 50 budgets. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted. | +| `verify-pass-binding` | **implemented** | A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite. | Verification (src/verify.js, forge verify) · master after 1.4.3 (unreleased) | [`src/verify.js`](../../src/verify.js)
[`test/verify.test.js`](../../test/verify.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Re-assessed on 7eef611 (the merge of the 2026-09-26 review fixes): the review's counterexamples F01 (renames, byte moves, empty files, modes, symlinks), F08 (a failing nested workspace) and F10 (code changed during the run) are regression tests, and a seeded property test moves the fingerprint under any composition of manifest transformations. Scope: gitignored files, declared `verify.generated` outputs and interpreter caches are not bound; a nested repository is bound by path only (listed as `unbound`); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted. | +| `ledger-evidence-independence` | **partial** | A claim's confidence rises only with independent evidence for that claim. | Ledger (src/ledger_store.js, src/ledger_bridge.js) · master after 1.4.3 (unreleased) | [`src/ledger.js`](../../src/ledger.js)
[`src/ledger_store.js`](../../src/ledger_store.js)
[`src/ledger_bridge.js`](../../src/ledger_bridge.js)
[`test/ledger_store.test.js`](../../test/ledger_store.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: abbreviations of one git object count as one event, a replayed or re-cited ref cannot refresh decay, and a non-equivalent rewrite no longer inherits trust (F06, F07 regression tests; a seeded property test over spelling mixes, replays and order). Not yet held: two different references to the same underlying run (for example a CI run id and its commit) still count as two events. At d2abfa6 this claim was refuted. | +| `reuse-exact-identity` | **implemented** | An exact reuse hit serves an artifact minted for the same specification (up to whitespace and Unicode normalization), and only while its bytes still match. | Reuse cache (src/reuse.js) · master after 1.4.3 (unreleased) | [`src/reuse.js`](../../src/reuse.js)
[`test/reuse.test.js`](../../test/reuse.test.js)
[`test/trust_properties.test.js`](../../test/trust_properties.test.js) | Re-assessed on 7eef611: F04 (operator, case, literal and polarity pairs) and F05 (edited or deleted artifacts) are regression tests, and a seeded property test never serves an artifact after an edit, move or deletion. Bytes are checked at serve time wherever the repository root is known (`forge reuse query`, the substrate); dependency contracts need an atlas, and without one a hit is marked requiresRevalidation. Artifacts minted before key version 2 never exact-hit. At d2abfa6 this claim was refuted. | +| `imagine-sandbox` | **refuted** | forge imagine --run executes the selected tests in a sandbox. | Consequence simulation (src/imagine.js, forge imagine) · master after 1.4.3 (unreleased) | [`src/imagine.js`](../../src/imagine.js)
[`docs/GUIDE.md`](../../docs/GUIDE.md) | Still false at 7eef611, and no longer claimed: it is an isolated git checkout (a detached-HEAD worktree) that isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. The CLI and docs now say "an isolated checkout of HEAD, not a security sandbox", and a suite that is not node:test gets an explicit unsupported-runner result. | | `integrations-emission` | **implemented** | Forgekit emits native config for ten coding tools plus MCP config for Roo Code and VS Code. | Config compiler (src/sync.js, src/emit) · 1.4.3 | [`docs/INTEGRATIONS.md`](../../docs/INTEGRATIONS.md)
[`test/sync.test.js`](../../test/sync.test.js)
[`test/mcp.test.js`](../../test/mcp.test.js) | Emission is tested; no host tool is launched in CI. Automatic hooks and in-agent enforcement exist only on Claude Code; a git pre-commit gate covers any tool that commits through git. | | `pcm-evidence-referenced-memory` | **implemented** | Proof-carrying memory stores content-addressed claims that carry references to their evidence. | Ledger (src/ledger.js) · 1.4.3 | [`docs/adr/0006-proof-carrying-memory.md`](../../docs/adr/0006-proof-carrying-memory.md)
[`src/ledger.js`](../../src/ledger.js) | A name, not a formal proof: there is no theorem prover in the loop. | | `theorem-d-joint-maxima` | **refuted** | Instructions and deterministic checks together can reach a residual of (1 - p_max)(1 - q_max). | Formal synthesis, Theorem D · HTML edition, corrected 2026-09-26 | [`research/formal-synthesis/substrate_synthesis.html`](../../research/formal-synthesis/substrate_synthesis.html)
[`research/empirical-refutation/extended_preprint.html`](../../research/empirical-refutation/extended_preprint.html)
[`research/recompute_corrections.py`](../../research/recompute_corrections.py) | Holds only if both maxima are attainable under one policy. Policy A (0.5, 0.9) leaves 0.05 and policy B (0.9, 0.1) leaves 0.09, while the separate maxima suggest 0.01; the attainable residual is a minimum over the joint feasible set, and the product is a lower bound. Asserted by recompute_corrections.py --theorem-checks. | diff --git a/docs/status/claims.json b/docs/status/claims.json index ea1dfc21..02fcb742 100644 --- a/docs/status/claims.json +++ b/docs/status/claims.json @@ -80,16 +80,16 @@ }, { "id": "node-impact-fixture-quality", - "claim": "forge impact scores precision 0.17, recall 1.00, F1 0.29 on six hand-labelled cases from this repository.", + "claim": "forge impact scores precision 0.17, recall 1.00, F1 0.28 on six hand-labelled cases from this repository.", "component": "Node code graph (src/atlas.js, forge impact)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "measured", "evidence": [ "reports/benchmarks.md", "bench/impact_cases.mjs" ], - "notes": "Six self-labelled symbols in one repository; the transitive walk is scored against direct-only labels. The 2026-09-26 review measured macro 0.18 / 1.00 / 0.30 on the same cases. A regex-derived graph, not the evaluated Python AST oracle; none of the Python study's numbers apply to it." + "notes": "Six self-labelled symbols in one repository; the transitive walk is scored against direct-only labels. Re-measured on the tree merged as 7eef611, after the CLI handler move added src/cli/memory.js as an eleventh claimText dependent (reports/benchmarks.md: mean F1 0.28, was 0.29). The 2026-09-26 review measured macro 0.18 / 1.00 / 0.30 on the same cases at d2abfa6. A regex-derived graph, not the evaluated Python AST oracle; none of the Python study's numbers apply to it." }, { "id": "router-gate-demo-saving", @@ -247,54 +247,55 @@ "id": "universal-router-target-guarantee", "claim": "The target:p objective guarantees a success rate of at least p.", "component": "Universal router (src/router, forge route universal)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "refuted", "evidence": [ "docs/UNIVERSAL_ROUTING.md", "bench/universal-router/README.md" ], - "notes": "Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist." + "notes": "Predicted cascade success was optimistic by 4-6 points on the run-4 test split (repository-reported), so target:p lands below p; in the in-repo replay (match-best-single) predicted minus observed success ranged from -2.8 to +9.1 points across six splits. Not a guarantee until an out-of-fold calibration map, reliability intervals and an abstention policy exist. Unchanged at 7eef611." }, { "id": "universal-router-budget-contract", "claim": "The budget:B objective returns a cascade whose expected cost is at most B, and says so explicitly when none exists.", "component": "Universal router (src/router, forge route universal)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "partial", "evidence": [ "docs/UNIVERSAL_ROUTING.md", "src/router/policy.js" ], - "notes": "B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). At the assessed commit an infeasible budget returned the least expensive cascade with no budgetMet:false (review F12); aedddf5 (committed after the assessed commit) reports feasible:false, budgetMet:false, minimumExpectedCost, maxPossibleCost and a reason, and the CLI prints INFEASIBLE." + "notes": "B bounds expected cost, not spend: a cascade can cost the sum of all its attempts, and E[cost] is under-predicted by 5-22% in the in-repo replay (see universal-router-cascade-cost-independence). The explicit half now holds at 7eef611: an infeasible budget returns ok:false, feasible:false, budgetMet:false, minimumExpectedCost and a reason (the CLI prints INFEASIBLE), with the least-bad cascade only as a labeled fallback, and every recommendation reports maxPossibleCost (review F12, regression-tested)." }, { "id": "universal-router-cascade-cost-independence", "claim": "A later cascade attempt costs the same in expectation whether or not earlier attempts failed.", "component": "Universal router cost model (src/router/policy.js)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "refuted", "evidence": [ "bench/universal-router/README.md", "bench/universal-router/holdout_eval.mjs", "docs/UNIVERSAL_ROUTING.md" ], - "notes": "The cascade cost formula assumes E[cost_i | earlier attempts failed, x] = E[cost_i | x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes." + "notes": "The cascade cost formula assumes E[cost_i | earlier attempts failed, x] = E[cost_i | x]. On the recorded SWE-bench Verified runs (in-repo replay, holdout_eval.mjs) expected cascade cost was under-predicted in all six splits, by 5-22% of the replayed cost ($0.102 expected against $0.124 replayed per task for seed 20260926): a failed attempt costs on average 1.2-2.0 times a successful one, for every model, and a later attempt is reached only after a failure. Corrected wording: this formula's E[cost] is optimistic until the cost model conditions on earlier outcomes. Unchanged at 7eef611." }, { "id": "route-outcome-provenance", "claim": "Outcomes recorded with forge route outcome are verified results.", "component": "Universal router outcome log (src/router/index.js)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", - "status": "refuted", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "status": "partial", "evidence": [ - "docs/UNIVERSAL_ROUTING.md", - "src/router/index.js" + "src/router/index.js", + "test/router_universal.test.js", + "docs/UNIVERSAL_ROUTING.md" ], - "notes": "At the assessed commit the pass/fail was caller-entered, with no provenance or attempt identity, so replaying a log multiplied evidence. aedddf5 (committed after the assessed commit) labels records provenance: self-reported unless --verify-run names a forge verify run whose verdict agrees (then verify-event), makes recording idempotent by attemptId (--attempt), and schema-validates records on write and read. Re-assess before changing the status." + "notes": "Re-assessed on 7eef611: records are schema-validated on write and read, recording is idempotent by attemptId (--attempt), and a row is provenance verify-event only when --verify-run names a forge verify run whose verdict agrees. Without --verify-run a row is labeled self-reported, which is the default, so the claim holds only for rows recorded that way. At d2abfa6 it was refuted (caller-entered, replayable)." }, { "id": "cost-reduction-90-target", @@ -355,29 +356,30 @@ "id": "phase-p3-reuse-cache", "claim": "P3: the reuse cache serves an artifact only while its evidence holds, and refuses a stale one.", "component": "Reuse cache (src/reuse.js, forge reuse)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "implemented", "evidence": [ "docs/plans/substrate-v2/00-overview.md", "src/reuse.js", - "test/reuse.test.js" + "test/reuse.test.js", + "test/trust_properties.test.js" ], - "notes": "Acceptance counterexamples at the assessed commit: exact keys erased operators and case (F04), and changed or deleted artifacts still exact-hit (F05). aedddf5 (committed after the assessed commit) repairs both with regression tests; re-assess before relying on it, and mark this phase partial if the probes still fail." + "notes": "Re-assessed on 7eef611: the acceptance counterexamples found at d2abfa6, exact keys that erased operators and case (F04) and changed or deleted artifacts that still exact-hit (F05), are repaired with regression tests and a seeded property test." }, { "id": "phase-p4-context-assembly", "claim": "P4: context assembly never exceeds its token budget and reports a computed missing set.", "component": "Context assembly (src/context.js, forge context)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "partial", "evidence": [ "docs/plans/substrate-v2/04-context-assembly.md", "docs/plans/substrate-v2/00-overview.md", "src/context.js" ], - "notes": "Over budget and pointer-only coverage were reported as complete (F02, F03). aedddf5 (committed after the assessed commit) reports overflow and pending reads; tokens are a chars/3.6 estimate of the rendered block; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered." + "notes": "Re-assessed on 7eef611: the rendered block never exceeds the budget (a seeded property test over 50 budgets), overflow and pending reads are reported, and a pointer is not coverage (F02, F03). Still partial: tokens are a chars/3.6 estimate; selection is a value-density heuristic with no approximation guarantee; the ambient hook does not assemble context; ok means syntactically delivered." }, { "id": "phase-p5-loop-closure", @@ -437,69 +439,76 @@ }, { "id": "context-completeness", - "claim": "forge context COMPLETE means the required knowledge was delivered within the token budget.", + "claim": "forge context COMPLETE means every required item was delivered as content within the (estimated) token budget.", "component": "Context assembly (src/context.js, forge context)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", - "status": "refuted", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "status": "implemented", "evidence": [ - "docs/plans/substrate-v2/04-context-assembly.md", - "docs/GUIDE.md" + "src/context.js", + "test/context.test.js", + "test/trust_properties.test.js", + "docs/plans/substrate-v2/04-context-assembly.md" ], - "notes": "At the assessed commit a 1-token budget returned ok:true with 39 tokens (F02), and a one-line pointer counted as delivering a definition below line 100 (F03). The contract repaired in aedddf5 (committed after the assessed commit): ok means syntactically delivered, overflow and pending reads are reported, and semantic sufficiency is not measured." + "notes": "Re-assessed on 7eef611: F02 (over budget reported as ok) and F03 (a pointer counted as coverage) are regression tests, and a seeded property test holds the rendered block within budget for 50 budgets. Tokens are a chars/3.6 estimate of the rendered block, not a tokenizer count; COMPLETE means syntactically delivered, and whether the content suffices for the edit is not measured. At d2abfa6 the stronger wording was refuted." }, { "id": "verify-pass-binding", - "claim": "A forge verify PASS is bound to the exact code state that was tested and covers every declared suite.", + "claim": "A forge verify PASS is bound to the code state that was tested (HEAD, staged and unstaged diffs, untracked non-ignored files) and covers every declared suite.", "component": "Verification (src/verify.js, forge verify)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", - "status": "refuted", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "status": "implemented", "evidence": [ "src/verify.js", "test/verify.test.js", + "test/trust_properties.test.js", "docs/GUIDE.md" ], - "notes": "Counterexamples at the assessed commit: the working-tree fingerprint ignored paths and file boundaries (F01), a failing nested workspace still gave a whole-repo PASS (F08), and code changed during the run was signed (F10). aedddf5 (committed after the assessed commit) uses a length-delimited manifest, reports uncovered workspaces as INCOMPLETE and detects mutation; re-run the review's probes before changing this status." + "notes": "Re-assessed on 7eef611 (the merge of the 2026-09-26 review fixes): the review's counterexamples F01 (renames, byte moves, empty files, modes, symlinks), F08 (a failing nested workspace) and F10 (code changed during the run) are regression tests, and a seeded property test moves the fingerprint under any composition of manifest transformations. Scope: gitignored files, declared `verify.generated` outputs and interpreter caches are not bound; a nested repository is bound by path only (listed as `unbound`); an unreadable untracked file makes the state unbindable. At d2abfa6 this claim was refuted." }, { "id": "ledger-evidence-independence", "claim": "A claim's confidence rises only with independent evidence for that claim.", "component": "Ledger (src/ledger_store.js, src/ledger_bridge.js)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", - "status": "refuted", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "status": "partial", "evidence": [ + "src/ledger.js", "src/ledger_store.js", - "src/ledger_bridge.js" + "src/ledger_bridge.js", + "test/ledger_store.test.js", + "test/trust_properties.test.js" ], - "notes": "At the assessed commit four prefixes of one commit raised confidence past the 0.8 threshold (F06), and a contradictory rewrite kept its evidence and 0.64 confidence (F07). aedddf5 (committed after the assessed commit) repairs this with regression tests; re-run the review's probes before changing the status." + "notes": "Re-assessed on 7eef611: abbreviations of one git object count as one event, a replayed or re-cited ref cannot refresh decay, and a non-equivalent rewrite no longer inherits trust (F06, F07 regression tests; a seeded property test over spelling mixes, replays and order). Not yet held: two different references to the same underlying run (for example a CI run id and its commit) still count as two events. At d2abfa6 this claim was refuted." }, { "id": "reuse-exact-identity", - "claim": "An exact reuse hit serves an artifact minted for the same specification, and only while its bytes still match.", + "claim": "An exact reuse hit serves an artifact minted for the same specification (up to whitespace and Unicode normalization), and only while its bytes still match.", "component": "Reuse cache (src/reuse.js)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", - "status": "refuted", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", + "status": "implemented", "evidence": [ "src/reuse.js", - "test/reuse.test.js" + "test/reuse.test.js", + "test/trust_properties.test.js" ], - "notes": "At the assessed commit '>= 18' and '<= 18' shared an exact key (F04), and an edited or deleted artifact still exact-hit (F05). aedddf5 (committed after the assessed commit) repairs this with regression tests; re-run the review's probes before changing the status." + "notes": "Re-assessed on 7eef611: F04 (operator, case, literal and polarity pairs) and F05 (edited or deleted artifacts) are regression tests, and a seeded property test never serves an artifact after an edit, move or deletion. Bytes are checked at serve time wherever the repository root is known (`forge reuse query`, the substrate); dependency contracts need an atlas, and without one a hit is marked requiresRevalidation. Artifacts minted before key version 2 never exact-hit. At d2abfa6 this claim was refuted." }, { "id": "imagine-sandbox", "claim": "forge imagine --run executes the selected tests in a sandbox.", "component": "Consequence simulation (src/imagine.js, forge imagine)", - "version": "1.4.3", - "source_commit": "d2abfa69fb77531199ffc67c5c076b524af69040", + "version": "master after 1.4.3 (unreleased)", + "source_commit": "7eef61179d16f6cdf8b5e929ce35a023aa17d1b0", "status": "refuted", "evidence": [ "src/imagine.js", "docs/GUIDE.md" ], - "notes": "It is an isolated git checkout (a detached-HEAD worktree): it isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. Since aedddf5 (committed after the assessed commit) the CLI says so (\"an isolated checkout of HEAD; not a security sandbox\"), and a suite that is not node:test gets an explicit unsupported-runner result." + "notes": "Still false at 7eef611, and no longer claimed: it is an isolated git checkout (a detached-HEAD worktree) that isolates checkout files, not network, credentials, the home directory or process permissions, and it tests the committed baseline, not uncommitted patches. The CLI and docs now say \"an isolated checkout of HEAD, not a security sandbox\", and a suite that is not node:test gets an explicit unsupported-runner result." }, { "id": "integrations-emission", diff --git a/landing/index.html b/landing/index.html index 161531f7..c4cc829a 100644 --- a/landing/index.html +++ b/landing/index.html @@ -975,7 +975,7 @@ .tool-grid { display: grid; - grid-template-columns: repeat(3, 1fr); + grid-template-columns: repeat(5, 1fr); margin: clamp(4rem, 7vw, 7rem) 0 0; padding: 0; border-top: 1px solid var(--line-strong); @@ -983,6 +983,13 @@ list-style: none; } +/* Ten tools: two full rows of five on wide screens, two columns below that. */ +@media (max-width: 1180px) { + .tool-grid { + grid-template-columns: repeat(2, 1fr); + } +} + .tool-grid li { display: grid; grid-template-columns: 30px 48px 1fr; @@ -1783,7 +1790,7 @@ -
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory and foresight—with automatic guardrails on Claude Code—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceNine native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
851ms
End-to-end benchmark
Blast-radius scan
1.68ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
+
Open source cognitive substrateforgekit v1.4.3 · beta

One operating
memory. Every
coding agent.

ForgeKit gives every AI coding tool the same memory and foresight—with automatic guardrails on Claude Code—without locking your work inside one vendor or one chat window.

Runtime deps
0
Native targets
9
License
MIT
FK / PREFLIGHTSYSTEM READY
01
REQUESTRefactor authentication flow
00:118
  1. 01Memory recalledPASS
  2. 02Blast radius mappedPASS
  3. 03Guardrails checkedPASS
TRACE FK-031-7D4PROCEED →
01 / The substrateState before action

The missing layer between
your intent and your agent.

Models are capable. Their operating context is fragile. ForgeKit supplies the durable layer that travels with the repository and shows up before the next action.

ACTIVE CAPABILITY / 01

Context that survives the chat.

Forge keeps decisions, lessons, and project state in the repository—so Claude, Codex, Cursor, and the next agent all inherit the same working memory.

3 records recalled
TYPERECORDSTATE
decisionUse SQLite for local-first state94%
lessonRun schema checks before generation88%
preferenceKeep the CLI dependency-free82%
02 / The protocolOne request · five checks · one trace

Action should leave evidence.

Forge turns agent behavior into a reviewable sequence. Each meaningful move begins with context and ends with proof.

  1. 01Recall

    Load relevant decisions and lessons.

  2. 02Classify

    Measure scope, cost, and reversibility.

  3. 03Foresee

    Map downstream surfaces before editing.

  4. 04Gate

    Pause risky or under-specified actions.

  5. 05Trace

    Record what changed and how it was verified.

03 / One sourceTen native targets

Change the agent. Keep the operating system.

One source emits each tool’s native configuration. Your rules and memory stay with the project—not the provider.

  • 01Claude Code
  • 02Codex
  • 03Cursor
  • 04Gemini
  • 05Aider
  • 06Copilot
  • 07Windsurf
  • 08Zed
  • 09Continue
  • 10OpenClaw

Plus MCP configuration for Roo Code and VS Code-compatible clients.

04 / Evidence ledgerMeasured, not invented

Fast enough to stay in the loop.

ForgeKit publishes the measurements behind its claims. The numbers below come from repository benchmarks and evaluation reports—not a marketing dashboard.

Pre-action gate
851ms
End-to-end benchmark
Blast-radius scan
1.68ms
Heuristic analysis
Held-out routing cost
+20.2%
vs always-premium, 80 tasks
Runtime dependencies
0
Node.js standard library
Inspect the evidence
05 / Honest limitsProfessional, not magical

The guardrail is not the road.

ForgeKit improves agent judgment; it does not replace yours. The project labels its assumptions so you can decide where to trust, test, or intervene.

  • 01

    Claude Code is the deepest-tested integration. Other targets have less real-world exercise today.

  • 02

    Blast-radius analysis is heuristic. It guides review; it is not a formal dependency proof.

  • 03

    Guardrails are not a sandbox. Keep permissions, review, and backups appropriate to the work.

06 / Start hereAbout sixty seconds

Give the next agent a better starting point.

Install ForgeKit, run forge init in your repository, and keep one shared operating context across every tool.

Open the quickstart
forgekit / install
 /plugin marketplace add CodeWithJuber/forgekit
  /plugin install forgekit
 

Recommended · ambient guards on every prompt

diff --git a/mintlify/cli/config.mdx b/mintlify/cli/config.mdx index 344c0ccf..d1f311af 100644 --- a/mintlify/cli/config.mdx +++ b/mintlify/cli/config.mdx @@ -51,9 +51,16 @@ forge models --json # the full resolution ## `forge dash` Local dashboard over the ledger, metrics, and blast radius. Read-mostly: the only writes -are the two human-driven actions `POST /api/ratify` and `POST /api/retract`, which are -guarded against CSRF and DNS-rebinding — a request must come from the loopback `Host` (and, -for a browser, a loopback `Origin`) or it is refused with `403`. +are the two human-driven actions `POST /api/ratify` and `POST /api/retract`. + +- **Every route checks `Host`.** Reads included, a request whose `Host` is not this loopback + address and port is refused with `403`, which blocks DNS rebinding. +- **Writes need the page's session token.** The server mints a random token at start-up, + embeds it in the page it serves, and requires it back in the `x-forge-token` header. A + browser `Origin` must be exactly this server's origin, so a page on another localhost port + cannot write. A script that POSTs must read the token from the page first. +- **Spend is labeled.** Missing session logs render as "spend unknown", never `$0`, and a + model without a price is "unpriced". ```bash forge dash # localhost-only (default port 4242) @@ -94,6 +101,11 @@ Reads `package.json`, `pyproject.toml`, `go.mod`, `Cargo.toml`, `Gemfile`, frameworks, package managers, and the repo's **actual** test commands — which feed the substrate's verification checklist. +`test` is what the repo declares it runs: an explicit `scripts.test` wins, and npm's +`"no test specified"` placeholder is not a suite. A runner that is installed but not what +the repo runs (a devDependency like `vitest`) is listed as `available`: inventory, not an +obligation `forge verify` will demand. + ## `forge report` Generate a static HTML report of the repo's Forge state — the ledger, metrics, and blast diff --git a/mintlify/cli/memory.mdx b/mintlify/cli/memory.mdx index e715767a..1f9c792f 100644 --- a/mintlify/cli/memory.mdx +++ b/mintlify/cli/memory.mdx @@ -50,6 +50,8 @@ Proof-carrying memory — the content-addressed claim store. ```bash forge ledger stats # what the repo knows, by kind and trust level forge ledger verify # re-check claims are in normal form +forge ledger verify --fix [--dry-run] # re-address old claims (--dry-run previews, writes nothing) +forge ledger compact [--dry-run] # archive what this ledger's own history says is unused forge ledger show # a claim and its evidence forge ledger blame # who minted it, every oracle outcome, per-author trust forge ledger query "" # retrieve by relevance @@ -64,6 +66,15 @@ forge ledger import # back-fill pre-ledger history into the ledge Add `--personal` for the per-user ledger. +Evidence counts once per event: four spellings of one commit (`git:` plus 7, 8, 9 or 40 +characters) are one vote, and a resolvable abbreviation is stored under the full id. +A reworded lesson inherits its predecessor's evidence only when the rewrite is equivalent up +to case, whitespace and punctuation; otherwise it starts at the prior and records what it +supersedes. `compact` merges near-duplicates only when they also agree on operators, +numbers, literals, identifiers, paths and negation. "Enable X" and "Disable X" are listed as +conflicts for a human to resolve, never merged. Every archived claim records why it went to +the attic (`tombstoned`, `dormant`, `idle` or `duplicate`); archived is not refuted. + ## `forge reuse` Proof-carrying code cache — served only when its evidence still holds. @@ -74,6 +85,15 @@ forge reuse mint "" --file # add an artifact to the cache forge reuse stats # cache stats ``` +- **Exact means the same text.** The key ignores only whitespace; case, operators, + literals and punctuation all count, so `age >= 18` and `age <= 18` never share a key. +- **Near must also agree on behaviour.** A reworded match whose operators, numbers, + literals, identifiers, paths or negation differ drops to `adapt`, with a note naming the + difference. +- **Checked where it is served.** An artifact whose file changed or was deleted since it + was minted, or whose dependency's declaration changed, is not served. Without an atlas to + check against, a hit is marked `NOT revalidated` (`requiresRevalidation: true`). + ## `forge handoff` Bounded session snapshot — rewrites `.forge/state.md`, re-injected each session start. diff --git a/mintlify/cli/quality.mdx b/mintlify/cli/quality.mdx index 1fd0c914..37e5098c 100644 --- a/mintlify/cli/quality.mdx +++ b/mintlify/cli/quality.mdx @@ -28,6 +28,28 @@ Only `PASS` counts as verified. `--deep` deliberately refuses to promote a conse `PASS` when the underlying `forge verify` tests status is anything other than `PASS`, so a green deep run always implies a green base run. +**What a `PASS` covers.** `forge verify` plans one suite per package that declares its own +tests (the root, plus each nested package with an explicit `scripts.test`, a pytest config, +a `go.mod`, …) and runs each in its own directory. `packages: n/m covered` counts the ones +that reached a verdict. A package whose suite never reached one makes the result +`INCOMPLETE`, and a failing package makes it `FAIL`, however green the root is. A root +script that already runs every workspace (`npm test --workspaces`, `pnpm -r test`, +`turbo run test`, …) covers them in one run. Tune it under `verify` in +`.forge/forge.config.json`: + +```json +{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"] } } +``` + +**Bound to the code that was tested.** The stamp is bound to a fingerprint of the working +tree (HEAD, staged and unstaged diffs, and each untracked file's path, mode, size and +content hash), taken before **and** after the run. If the code changed while the tests ran +(a formatter, a generator, another agent), the result is `INCOMPLETE` with +`mutated: true`. Interpreter caches and the `generated` paths never count as a change. +Each run also appends a MAC-sealed event to `.forge/verify-events.jsonl` (run id, verifier +version, suites, coverage, pre/post state), which `forge route outcome --verify-run ` +can cite. + ## `forge scan` Skill-gate — vet a skill or MCP server for injection / RCE / exfil before install. diff --git a/mintlify/cli/substrate.mdx b/mintlify/cli/substrate.mdx index eb8439b0..1bc0936e 100644 --- a/mintlify/cli/substrate.mdx +++ b/mintlify/cli/substrate.mdx @@ -33,8 +33,16 @@ Recommend the cheapest capable model for a task. ```bash forge route "" forge route gateway # emit LiteLLM gateway config +forge route universal "" [--objective match-best-single|target:

|value:<$>|budget:<$>] [--provider ] +forge route outcome "" --model --pass|--fail [--cost ] [--attempt ] [--verify-run ] +forge route fit # refit the universal router on your recorded outcomes +forge route models # the registry: who serves which model, at what price ``` +`route universal` is the opt-in cross-provider router; see +[Model routing](/concepts/model-routing) +for what its numbers mean and what they do not. + ## `forge impact` Hazard-aware blast radius — SCC-aware propagation and data-driven threshold from @@ -81,10 +89,23 @@ Budgeted context assembly + completeness gate — what an edit NEEDS known. ```bash forge context "" +forge context "" --budget 4000 # tighter token budget +forge context "" --block # print the assembled context itself ``` -Assembles a budgeted context via set-cover over the predicted edit set, applies a -compression ladder, and reports the computed missing set. +Pins the required-knowledge set for the edit, downgrades items along a compression ladder +before dropping anything, fills the rest of the budget with a value-density heuristic, and +reports the computed missing set. + +- **`COMPLETE` means delivered, not sufficient.** Every required item arrived as content + inside the budget; whether that content is enough for the edit is not measured. +- **Token counts are an estimate** (characters ÷ 3.6 of the rendered block, labels and + separators included), not a tokenizer's count, and never exceed `--budget` while the + result reports success. +- **A pointer is not coverage.** An item that only fits as a `- read ` pointer is a + `pending` read; a 25-line head covers a definition only when the definition is inside it. +- **Over budget is `INCOMPLETE`.** When even pointers do not fit, the result is + `overflow: true`, `ok: false`, never a silent over-budget pass. ## `forge anchor` diff --git a/mintlify/concepts/model-routing.mdx b/mintlify/concepts/model-routing.mdx index 8f0e9dc2..aebe7250 100644 --- a/mintlify/concepts/model-routing.mdx +++ b/mintlify/concepts/model-routing.mdx @@ -93,6 +93,44 @@ name decide. `forge doctor`'s **gateway models** row prints the resolved `tier → model` mapping for verification. +## Universal router — `forge route universal` + +A separate, opt-in router: it recommends one model, or a cascade that escalates on a failed +check, across every provider in the registry, minimising **expected** cost for the success +you ask for. + +```bash +forge route universal "" # match the best single model +forge route universal "" --objective target:0.8 # or value:<$>, budget:<$> +forge route universal "" --provider anthropic # only models you can call there +``` + +- **Infeasible is explicit.** A budget no cascade fits, or a target none reaches, prints + `INFEASIBLE` and exits 1 (`ok: false`, `feasible: false`, the minimum achievable expected + cost, and the least-bad cascade only as a labeled `fallback` in `--json`). `budget:` bounds + expected cost, not what one task can spend; each recommendation also shows the cost if + every attempt runs. +- **In the registry is not callable.** A recommended model that no configured provider + serves is marked `no provider id` and the recommendation is `advice only` + (`applicable: false`). Add its id under `providers` in `.forge/models.json`, or pass + `--provider`. +- **Learning from your outcomes.** `forge route outcome` records one attempt, validated + (known model, finite non-negative cost, the right feature length). It is self-reported + unless `--verify-run ` ties it to a matching `forge verify` run, and + `--attempt ` makes re-recording idempotent. `forge route fit` then updates the shipped + prior with them. + + + The shipped prior was fitted on public SWE-bench Verified outcomes: one agent scaffold, + Python repositories, February 2026 prices. Its refit reproduces exactly from pinned public + data (`bench/universal-router/reproduce.sh`). The 76.3% / $0.093 held-out headline is + repository-reported from an external harness, not independently reproduced. In an in-repo + held-out replay the router did not beat a fixed cascade on solve rate, and expected cost was + under-predicted by 5–22% because failed attempts cost more. Treat it as a starting point for + your own outcomes, not a guarantee. The full evidence and limits are in + [docs/UNIVERSAL_ROUTING.md](https://github.com/CodeWithJuber/forgekit/blob/HEAD/docs/UNIVERSAL_ROUTING.md). + + ## Providers and cost ```bash @@ -103,5 +141,7 @@ forge cost --stages # measured per-stage cost factors `forge cost --stages` reports **measured stages only** — a stage with no events says - "no data", never a default. A number is an assumption until measured. + "no data", never a default. A number is an assumption until measured. A day or stage with + no logs is **unknown, not $0**; spend figures are estimates (logged tokens × list price, + USD), and a model without a price is "unpriced", not free. diff --git a/mintlify/concepts/proof-carrying-memory.mdx b/mintlify/concepts/proof-carrying-memory.mdx index 4a863b1f..b4be7738 100644 --- a/mintlify/concepts/proof-carrying-memory.mdx +++ b/mintlify/concepts/proof-carrying-memory.mdx @@ -84,6 +84,7 @@ for audit, never silently removed. ```bash forge ledger stats # what the repo knows, by kind and trust level forge ledger verify # re-check claims are in normal form +forge ledger verify --fix --dry-run # preview re-addressing old claims; writes nothing forge ledger show # a claim and its evidence trail forge ledger blame # who minted it, every oracle outcome, per-author trust forge ledger query "" # retrieve claims by relevance @@ -98,9 +99,16 @@ Add `--personal` for the per-user ledger. ## The reuse cache is proof-carrying too `forge reuse` is a proof-carrying code cache. A generated artifact is only served again -when its evidence still holds — the confidence is above the floor **and** its atlas -dependencies still resolve. Otherwise it falls through to generation and mints a fresh -claim on the way back. +when its evidence still holds. Its confidence must be above the floor, its file must be +byte-identical to what was verified, and its dependencies must still resolve with the same +declarations. Otherwise it falls through to generation and mints a fresh claim on the way +back. An exact hit needs the same text (only whitespace is ignored). A near hit must also +agree on operators, numbers, literals, identifiers, paths and negation, so `age >= 18` is +never served for `age <= 18`. + +One event is one vote. Abbreviations of one commit count once. A reworded lesson inherits +trust only when the rewrite is equivalent. Similar-but-opposite rules are reported as +conflicts, never merged. ```mermaid %%{init: {'theme':'base','themeVariables':{'primaryColor':'#201a15','primaryTextColor':'#f2ede7','primaryBorderColor':'#372c22','lineColor':'#f26430','secondaryColor':'#272019','tertiaryColor':'#171310','edgeLabelBackground':'#201a15','clusterBkg':'#171310','clusterBorder':'#4a3b2e','fontFamily':'ui-sans-serif, system-ui, sans-serif','fontSize':'14px'},'flowchart':{'curve':'basis','padding':10,'nodeSpacing':36,'rankSpacing':44}}}%% diff --git a/mintlify/concepts/verification-gates.mdx b/mintlify/concepts/verification-gates.mdx index 13557879..26422369 100644 --- a/mintlify/concepts/verification-gates.mdx +++ b/mintlify/concepts/verification-gates.mdx @@ -45,6 +45,28 @@ The `PASS`-implies-base-`PASS` rule is deliberate: a green deep consensus can ne outrun a red base run, so `INCOMPLETE` or `NOT_CONFIGURED` on the base test lens downgrades the deep result to match. +**What a `PASS` covers.** `forge verify` plans one suite per package that declares its own +tests (the root, plus each nested package with an explicit `scripts.test`, a pytest config, +a `go.mod`, …) and runs each in its own directory. `packages: n/m covered` counts the ones +that reached a verdict. A package whose suite never reached one makes the result +`INCOMPLETE`, and a failing package makes it `FAIL`, however green the root is. A root +script that already runs every workspace (`npm test --workspaces`, `pnpm -r test`, +`turbo run test`, …) covers them in one run. Tune it under `verify` in +`.forge/forge.config.json`: + +```json +{ "verify": { "workspaces": "auto", "exclude": ["packages/legacy"], "generated": ["coverage/**"] } } +``` + +**Bound to the code that was tested.** The stamp is bound to a fingerprint of the working +tree (HEAD, staged and unstaged diffs, and each untracked file's path, mode, size and +content hash), taken before **and** after the run. If the code changed while the tests ran +(a formatter, a generator, another agent), the result is `INCOMPLETE` with +`mutated: true`. Interpreter caches and the `generated` paths never count as a change. +Each run also appends a MAC-sealed event to `.forge/verify-events.jsonl` (run id, verifier +version, suites, coverage, pre/post state), which `forge route outcome --verify-run ` +can cite. + ## The hallucinated-symbol flag — `forge atlas has` `forge atlas has ` is the hallucination check: if the model calls a symbol that From a168879f73e3c84e7f4258f89b079ab507864423 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 22:22:50 +0000 Subject: [PATCH 4/4] fix(docs): compare generated blocks in LF so Windows checkouts are not stale The Windows CI job (Git Bash, core.autocrlf) failed "the repository's changelog page is current": the page checks out with CRLF while `forge docs render` emits LF, so the byte comparison could never match. `forge docs check` had the same fault for every strict block on a Windows checkout, and `forge docs render` wrote LF block lines into CRLF files. src/docs_render.js now reads each managed doc in LF (renderDocs and renderFile), compares in LF, and writes a changed file back in its own line endings. The page test normalizes the file it reads. New tests reproduce a CRLF checkout for both render paths; both fail against the old renderer. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01GVVG2VDETWsDxMu6MBWPz2 --- CHANGELOG.md | 7 +++++++ mintlify/changelog/overview.mdx | 6 +++++- src/docs_render.js | 27 +++++++++++++++++++-------- test/changelog_page.test.js | 3 ++- test/docs_render.test.js | 27 +++++++++++++++++++++++++++ 5 files changed, 60 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index adc69368..01e06820 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -15,6 +15,13 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). high, pre-existing), the workspace-glob trim and the semantic guard's edge-punctuation trim now use linear scans with identical results (checked against the old regexes). +### Fixed + +- **`forge docs check` and `forge docs render` work on a Windows checkout.** With + `core.autocrlf`, Markdown checks out with CRLF line endings while generated blocks render + with LF, so every generated block read as stale and a render wrote LF lines into a CRLF + file. Blocks are now compared in LF and written back in the file's own line endings. + ### Added - **The docs site's changelog page is generated from `CHANGELOG.md`.** It had one hand-written diff --git a/mintlify/changelog/overview.mdx b/mintlify/changelog/overview.mdx index 74399ea3..b63939d0 100644 --- a/mintlify/changelog/overview.mdx +++ b/mintlify/changelog/overview.mdx @@ -14,12 +14,16 @@ This page is generated from `CHANGELOG.md` by `forge docs render`, and `forge do CI when it falls behind, so it cannot drift from the release notes again. {/* forge:render:changelog:begin (generated by `forge docs render` — do not edit) */} - + **Security** - **The claim-registry table escapes backslashes, and four patterns no longer backtrack quadratically on long runs (CodeQL).** +**Fixed** + +- **`forge docs check` and `forge docs render` work on a Windows checkout.** + **Added** - **The docs site's changelog page is generated from `CHANGELOG.md`.** diff --git a/src/docs_render.js b/src/docs_render.js index 09741642..2e7ba29c 100644 --- a/src/docs_render.js +++ b/src/docs_render.js @@ -222,6 +222,16 @@ function trackedMarkdown(root) { return out ? out.split("\n").filter(Boolean) : []; } +/** A Windows checkout (`core.autocrlf`) holds CRLF while blocks render in LF, so every + * strict block would read as stale and a write would mix line endings. Render and compare + * in LF; write back in the file's own line endings. */ +function readDoc(path) { + const raw = readFileSync(path, "utf8"); + const crlf = raw.includes("\r\n"); + return { text: crlf ? raw.replace(/\r\n/g, "\n") : raw, crlf }; +} +const withEol = (text, crlf) => (crlf ? text.replace(/\n/g, "\r\n") : text); + /** * Render every managed doc surface. With {write:true} stale files are rewritten; * otherwise this is a pure report (what `--check` and the docs-check reconciler use). @@ -231,18 +241,18 @@ function trackedMarkdown(root) { * missing:{file:string, name:string}[]}} */ export function renderDocs(root = BRAND.root, { write = false } = {}) { - /** @type {Map} */ + /** @type {Map} */ const touched = new Map(); const missing = []; const load = (file) => { if (!touched.has(file)) { - let text; + let doc; try { - text = readFileSync(join(root, file), "utf8"); + doc = readDoc(join(root, file)); } catch { return null; } - touched.set(file, { text, orig: text, strict: false, why: [] }); + touched.set(file, { text: doc.text, orig: doc.text, crlf: doc.crlf, strict: false, why: [] }); } return touched.get(file); }; @@ -291,7 +301,7 @@ export function renderDocs(root = BRAND.root, { write = false } = {}) { const files = []; for (const [file, doc] of touched) { const changed = doc.text !== doc.orig; - if (changed && write) writeFileSync(join(root, file), doc.text); + if (changed && write) writeFileSync(join(root, file), withEol(doc.text, doc.crlf)); if (changed || doc.why.length) files.push({ file, changed, strict: doc.strict, why: doc.why }); } // `missing` is informational (a root without markers manages nothing) — only STALE @@ -308,12 +318,13 @@ export function renderDocs(root = BRAND.root, { write = false } = {}) { * @returns {{found: boolean, changed: boolean}} */ export function renderFile(root, file, { write = false } = {}) { - let text; + let doc; try { - text = readFileSync(join(root, file), "utf8"); + doc = readDoc(join(root, file)); } catch { return { found: false, changed: false }; } + const { text } = doc; let next = text; let found = false; for (const t of BLOCK_TARGETS) { @@ -322,6 +333,6 @@ export function renderFile(root, file, { write = false } = {}) { found ||= r.found; next = r.text; } - if (write && next !== text) writeFileSync(join(root, file), next); + if (write && next !== text) writeFileSync(join(root, file), withEol(next, doc.crlf)); return { found, changed: next !== text }; } diff --git a/test/changelog_page.test.js b/test/changelog_page.test.js index 2dbfc9f0..1c0ad226 100644 --- a/test/changelog_page.test.js +++ b/test/changelog_page.test.js @@ -128,7 +128,8 @@ test("renderChangelogUpdates: one Update per non-empty release, tagged, linked t }); test("the repository's changelog page is current and MDX-safe", () => { - const page = readFileSync(join(BRAND.root, CHANGELOG_PAGE), "utf8"); + // A Windows checkout (core.autocrlf) holds CRLF; the renderer compares in LF. + const page = readFileSync(join(BRAND.root, CHANGELOG_PAGE), "utf8").replace(/\r\n/g, "\n"); const begin = page.indexOf("{/* forge:render:changelog:begin"); const end = page.indexOf("{/* forge:render:changelog:end */}"); assert.ok(begin !== -1 && end > begin, "the page carries the MDX render markers"); diff --git a/test/docs_render.test.js b/test/docs_render.test.js index 985ad076..0c73d632 100644 --- a/test/docs_render.test.js +++ b/test/docs_render.test.js @@ -116,6 +116,28 @@ test("renderDocs fills managed blocks, is idempotent, and reports tampering as s assert.ok(third.files.some((f) => f.strict && f.why.includes("block commands-table"))); }); +test("a CRLF checkout (Windows core.autocrlf) is current when its blocks are, and stays CRLF", () => { + const root = dir(); + const begin = ``; + writeFileSync( + join(root, "README.md"), + `# fixture\n\n${begin}\n\n`, + ); + renderDocs(root, { write: true }); + const crlf = readFileSync(join(root, "README.md"), "utf8").replace(/\n/g, "\r\n"); + writeFileSync(join(root, "README.md"), crlf); + const current = renderDocs(root, { write: false }); + assert.ok(current.ok, "line endings alone are not drift"); + assert.deepEqual(current.files, []); + // Real drift is still caught, and the rewrite keeps the file's own line endings. + writeFileSync(join(root, "README.md"), crlf.replace(`\`${BRAND.cli} rank\``, "`TAMPERED`")); + assert.ok(!renderDocs(root, { write: false }).ok); + renderDocs(root, { write: true }); + const after = readFileSync(join(root, "README.md"), "utf8"); + assert.ok(after.includes(`\`${BRAND.cli} rank\``)); + assert.ok(!/(^|[^\r])\n/.test(after), "no bare LF after a rewrite of a CRLF file"); +}); + test("renderDocs on a root without markers manages nothing and stays ok (fixture safety)", () => { const root = dir(); writeFileSync(join(root, "README.md"), "# plain fixture, hand-written table\n"); @@ -159,4 +181,9 @@ test("an MDX page's block uses JSX-comment markers; renderFile re-renders only t assert.match(text, /https:\/\/github\.com\/o\/r\/blob\/HEAD\/CHANGELOG\.md#100---2026-09-01/); assert.ok(renderDocs(root, { write: false }).ok, "rendered → current"); assert.equal(renderFile(root, "mintlify/changelog/overview.mdx").changed, false); + // The same files checked out with CRLF (Windows) are still current. + for (const f of [page, join(root, "CHANGELOG.md")]) + writeFileSync(f, readFileSync(f, "utf8").replace(/\n/g, "\r\n")); + assert.equal(renderFile(root, "mintlify/changelog/overview.mdx").changed, false); + assert.ok(renderDocs(root, { write: false }).ok, "CRLF alone is not drift"); });