From cd8a7bb959d4c33bc45cc8fe37d27f7a5de279b7 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Mon, 5 Oct 2026 22:26:11 +0000 Subject: [PATCH 1/7] docs(global): plan the rat-stack scorecard instrument Kiro approved this plan on 2026-10-05. It defines 29 metric families measured on both the starter and rat-stack, the verdict and ratchet rules, and the U1-U9 stack that builds them --- ...10-05-2151-feat-ratstack-scorecard-plan.md | 480 ++++++++++++++++++ 1 file changed, 480 insertions(+) create mode 100644 docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md diff --git a/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md b/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md new file mode 100644 index 0000000..60db791 --- /dev/null +++ b/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md @@ -0,0 +1,480 @@ +--- +title: rat-stack Scorecard - Plan +type: feat +date: 2026-10-05 +origin: docs/brainstorms/inputs/requirements-final.md +artifact_contract: ce-unified-plan/v1 +product_contract_source: legacy-requirements +execution: code +--- + +# rat-stack Scorecard - Plan + +## Goal Capsule + +- **Objective:** Kiro and Ryan can read, on every PR and on `main`, a table showing which rat-stack bins the starter beats. Each verdict is backed by numbers measured on both sides from running code, and comes with its provenance, so anyone can reproduce it. The same published JSON is what the site's later `/scorecard` page renders. +- **Means:** a verifier-owned Evaluator instrument in `evals/ratstack-scorecard/`. A zero-dependency Deno orchestrator drives sandboxed runs of both sides and publishes JSON plus a Markdown table from one CI workflow with a ratchet gate (KTD3, KTD12). +- **Authority:** Kiro rules scope and findings. `repos/constitution/` is the law. Requirements are in `docs/brainstorms/inputs/requirements-final.md` (Superiority Map), and the sandbox ruling is in `docs/brainstorms/inputs/ruling-nix-distribution-sandbox.md`. Builders never edit the instrument (R9). +- **Execution profile:** one `gh stack` on trunk `main`, branches `verify/scorecard-`, one PR per unit in U1-U9 order, each inert until the layer that wires it. The verifier session (`starter-verify`) builds and ships it; Kiro merges. +- **Stop conditions:** the same error three times, or a launcher capability that is missing (Dependencies). Either one stops the work and goes to Kiro with the exact error. No unsandboxed fallback, no off flag. + +--- + +## Product Contract + +### Summary + +The scorecard defines one or more metrics for each rat-stack bin and surface, runs both stacks to measure them, and records every number with its provenance. CI publishes the table as a job summary and the JSON as an artifact on every PR to `main` and every push to `main`. The job fails on instrument errors and on regressions against `main`. rat-stack runs as untrusted code inside the shared sandbox launcher. + +### Problem Frame + +The Superiority Map in `requirements-final.md` claims a checkable win over every rat-stack bin, but nothing checks it yet. Builders grading their own claims is the defect CONST-E9 names, and a number with no baseline or provenance certifies nothing. The starter on `main` today is one `hello` package, so almost every row starts as `absent`. The instrument has to report that honestly, then track each lake as it lands. + +### Key Decisions + +- **Local rows run the current rat-stack pin, and networked rows run live sites.** Local rows measure rat-stack at its pinned commit (main HEAD, `54d3560` today). Agent Readiness and home-page Lighthouse measure live `ratstack.sh` against the starter's deployed URL: the preview on PRs, production on `main`. Each networked row records the commit the live site serves and is flagged when that commit differs from the pin. (session-settled: user-directed — chosen over running the stale pin behind a tunnel, or scanning live only: "beat the CURRENT rat-stack, not a stale pin".) Governs R5, R6, R7. +- **rat-stack measurements are cached by rat-stack commit + instrument hash + nixpkgs rev, and the starter runs every time.** (session-settled: user-directed — chosen over measuring both sides on every PR.) Governs R12. +- **The gate fails on instrument errors, on starter regressions beyond the noise band, and on bins that lose beaten status.** (session-settled: user-directed — chosen over a report-only job; this answer is Kiro's GATE1 sign-off, 2026-10-05.) Governs R11. +- **The concurrency bin gets two race rows.** (a) The same workload on both sides wherever rat-stack has an analog: N concurrent redemptions of one single-use page ticket. (b) Each side's strongest invariant: the starter's 300 claims for 100 seats. rat-stack shows as `unsupported`, citing `packages/core/src/join-interest-contract.ts:22-26`, and counts as beaten only with that citation. (session-settled: user-directed — chosen over racing only each side's own invariant, or only the shared workload.) Governs R4, M11, M12. +- **Cloudflare URL Scanner `agentReadiness` is the primary readiness row, and isitagentready.com `/api/scan` is a cross-check.** When the two disagree, the row is flagged, not averaged. (session-settled: user-directed — chosen over isitagentready as primary: first-party, reproducible JSON through our token, matches R51's Agent Readiness 100.) Governs M1, M2. + +### Requirements + +**Measurement contract** + +- R1. Every rat-stack bin and surface in the origin's Superiority Map maps to at least one metric in the Metric Catalog, measured on both sides by running code. Reading documentation never produces a number. +- R2. A row is `beaten` only when the starter is strictly better and the comparison reproduces under KTD1. +- R3. Every value carries provenance: side, commit, instrument hash, nixpkgs rev, runner name, every run's raw value, UTC timestamp, and the versions of the tools that measured it. +- R4. Absence is explicit. A bin or surface the starter lacks is `absent`, with the reason. An invariant rat-stack lacks is `unsupported`, with a `file:line` citation that the instrument re-verifies at the pin. A row a side cannot be run for is `unmeasurable`, with the exact error, and Kiro rules on it. + +**Subjects** + +- R5. Local rows run rat-stack at the commit in `evals/ratstack-scorecard/ratstack.pin.json` and the starter at the commit under test. +- R6. Networked rows (M1-M3) measure `https://ratstack.sh` against the starter's deployed URL. They record the live commit each site serves and flag a rat-stack row whose live commit differs from the pin. +- R7. When rat-stack `main` moves past the pin, the scorecard opens a pin-bump PR in this repository. + +**Trust and ownership** + +- R8. Every process that loads third-party code runs in the shared sandbox launcher: both sides' dependencies, the instrument's npm tools, Claude Code, Lighthouse, and the debt counter. Only the orchestrator, which has no third-party imports, runs outside it. +- R9. `evals/ratstack-scorecard/**` and `.github/workflows/scorecard.yml` are verifier-owned Evaluator surfaces. The AGENTS.md line that records this goes to Kiro as a proposal in `docs/brainstorms/REFLECTION.md`. + +**Publication and gate** + +- R10. CI writes the table to `$GITHUB_STEP_SUMMARY` and uploads the JSON as an artifact on every PR to `main` and every push to `main`. The JSON validates against `evals/ratstack-scorecard/scorecard.schema.json`, the contract the site will render at `/scorecard`. +- R11. The scorecard job fails on any instrument error, on any starter metric regressing against `main` beyond its noise band (KTD2), and on any bin beaten on `main` that the PR no longer beats. +- R12. rat-stack values are reused from cache while the key from Key Decisions holds. The starter side is measured on every run. + +### Metric Catalog + +Each row names the rat-stack bin it grades, the unit, and the better direction. Kind `count` means three runs that must agree exactly. Kind `measurement` means N runs compared under KTD1. Every row runs on both sides. + +| ID | rat-stack bin / surface | Metric | Better | Kind / runs | Family | +| --- | ---------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------- | --------------- | -------------- | +| M1 | `apps/mischief` front door | URL Scanner `agentReadiness`: checks passed, plus level 0-5 | higher | count / 3 | networked | +| M2 | `apps/mischief` front door | isitagentready `/api/scan` level, as a cross-check against M1 | higher | count / 3 | networked | +| M3 | `apps/mischief` home page | Lighthouse on `/` with `Accept: text/html`: performance score, LCP ms, CLS, TBT ms, total byte weight, failed accessibility audits (one row each) | per sub-row | measurement / 5 | networked | +| M4 | `/pins.md` | Versions named on the served pins page that differ from the side's lockfile | lower | count / 3 | agent-surfaces | +| M5 | `/log.md` | Share of served change-log entries that link both a PR and a CI run | higher | count / 3 | agent-surfaces | +| M6 | `apps/web` | Gzipped client JS bytes emitted by the web build | lower | count / 3 | cold-path | +| M7 | `apps/mischief` Worker | Gzipped Worker script bytes emitted by the deploy build | lower | count / 3 | cold-path | +| M8 | `packages/capability` | Mean projections per read-only capability whose output for one input equals the HTTP output (CLI, HTTP, MCP, RPC, code mode, A2A, gRPC) | higher | count / 3 | agent-surfaces | +| M9 | `apps/cli` | Milliseconds from CLI process start to a correct answer for one read capability | lower | measurement / 5 | agent-surfaces | +| M10 | MCP + agent front door | Claude Code seconds to a correct answer through the side's MCP only: each of the 5 runs asks all 3 lockfile-derived questions, and the value is the median of the 15 answer times. Correctness rate is a sub-row | lower / higher | measurement / 5 | agent-surfaces | +| M11 | `packages/core` + `packages/intake-live` | Successes above 1 when 300 concurrent requests redeem one single-use token | lower | count / 3 | running-stack | +| M12 | `packages/core` lifecycles | Seats granted beyond capacity for 300 concurrent claims on 100 seats; rat-stack `unsupported` with citation | lower | count / 3 | running-stack | +| M13 | `packages/database` | Share of declared store adapters whose side's own store suite passes when that adapter is selected | higher | count / 3 | gate-mutation | +| M14 | `packages/auth` | Auth endpoints that accept a password credential (2xx on email + password sign-up or sign-in) | lower | count / 3 | running-stack | +| M15 | `packages/events` | Distinct cookies set on anonymous GETs of every sitemap page | lower | count / 3 | agent-surfaces | +| M16 | `packages/lore` | Operation × surface pairs (`search`, `read`, `backlinks`, `neighbors`, `mentions`, `path` × the seven surfaces) returning a schema-valid result | higher | count / 3 | agent-surfaces | +| M17 | `packages/subscriber-delivery` | Requests whose confirmations in the local mail sink are not exactly one, after the runtime is killed mid-burst (50 requests) and restarted | lower | count / 3 | running-stack | +| M18 | `packages/devtools` | Share of 20 scripted calls retrievable afterwards from the side's local devtools store with surface, capability and outcome | higher | count / 3 | agent-surfaces | +| M19 | `packages/intake-live` abuse bounds | Requests accepted beyond the side's own declared per-email bound, under 50 concurrent requests for one email | lower | count / 3 | running-stack | +| M20 | `packages/code-snippets` | Whether the build fails after a symbol quoted on a page is renamed (1 or 0) | higher | count / 3 | gate-mutation | +| M21 | `apps/infra` | Outbound connection attempts while importing the stack module, with no egress allowed | lower | count / 3 | gate-mutation | +| M22 | `apps/infra` | Whether removing one binding from the stack fails typecheck (1 or 0) | higher | count / 3 | gate-mutation | +| M23 | fence: debt ledger | Suppression directives in tracked source, counted by the neutral counter (KTD8) | lower | count / 3 | static | +| M24 | fence: tests | Sabotage kill rate: verifier sabotages of shared published behaviours that turn the side's own gate red (KTD9) | higher | count / 3 | gate-mutation | +| M25 | `keep-or-cut` / bin removal | Share of the side's documented bin removals after which its gate stays green | higher | count / 3 | gate-mutation | +| M26 | `acceptance-cold-clone.sh` | Seconds from fresh clone to green documented gate under KTD7 | lower | measurement / 3 | cold-path | +| M27 | install | HTTP requests during the cold install, counted by the recording proxy (KTD7), with hosts listed | lower | count / 3 | cold-path | +| M28 | install | Distinct `name@version` packages in the side's lockfile | lower | count / 3 | static | +| M29 | `skills/*` + `/.well-known/agent-skills` | Share of served skills that `skills add` (npm `skills` 1.7.0) installs and whose `SKILL.md` frontmatter validates | higher | count / 3 | agent-surfaces | + +Local rows run each side with its documented local defaults plus throwaway values for every declared secret that switches a feature on, for example rat-stack's `INTEREST_TOKEN_SECRET` and `EVENTS_ENABLED`. They never use real credentials. + +### Acceptance Examples + +- AE1. **Covers R2.** Given Lighthouse LCP runs of rat-stack [410, 420, 455, 430, 418] and starter [380, 395, 412, 390, 401], the row is `not-beaten`, because the starter's worst run (412) is not below rat-stack's best (410). +- AE2. **Covers R2.** Given M23 counts of 217, 217, 216 on one side, the row is `instrument-error`, because a count must reproduce exactly. +- AE3. **Covers R4, M12.** Given rat-stack at a pin where `join-interest-contract.ts:22-26` no longer contains the cited "Confirmation is not a seat" text, the row is `instrument-error` and not `beaten`. +- AE4. **Covers R11.** Given `main` beats rat-stack on M14 and a PR re-enables a password route, the scorecard job fails and names M14. +- AE5. **Covers R11.** Given `main`'s M9 runs [120, 131, 140, 125, 128] ms and the PR's median is 139 ms, the job passes, because 139 is inside the band. A PR median of 141 ms fails. +- AE6. **Covers R6.** Given `ratstack.sh/log.md` lists `ed63ba3` as its newest commit and the pin is `54d3560`, the M1-M3 rat-stack cells carry a `live≠pin` flag that shows in the table. +- AE7. **Covers R4.** Given a fork PR with no preview deployment, M1-M3 starter cells are `absent` with the reason "no deployment for ``", and the ratchet treats that as neutral. +- AE8. **Covers R8.** Given the rat-stack stack is running, a probe inside its sandbox fails to read `~/.ssh` and fails to connect to an undeclared host, and the scorecard records the probe in its provenance. + +### Scope Boundaries + +- Job 2 (stack review and QA on Kiro's prompt) is outside this plan. It runs per request and writes `docs/reviews/-findings.md`. +- The `/scorecard` page is the site's work in a later starter lake. This plan publishes only its data contract (R10). +- R51's absolute floor (Agent Readiness 100 on the PR preview) is the starter's own CI gate, owned by the lake that ships R51. The scorecard reports M1 and fails only per R11. Duplicating R51 here would be a second gate on the same number and would need its own GATE1 approval. +- The instrument never edits starter or rat-stack source in place. Sabotage and removal edits apply to throwaway clones. + +#### Outside this product's identity + +- Averaging disagreeing scanners, weighting rows into one composite score, or ranking bins by importance. Each row stands alone, as Kiro ruled for M1/M2. +- Running untrusted code outside the launcher for any reason, including speed. + +### Dependencies + +- D1. **Shared sandbox launcher** (`packages..sandbox`, branch `prm/nix-packaging` of systemfsoftware/pnpm-release-management, at `180122866dd537fa728b5563fb1820fbd2af88cc` on 2026-10-05, pinned by rev while its PR is open). It already provides `--allow-host HOST[:PORT]` (HTTPS `CONNECT` egress through its allow-list proxy, which can also target a host-loopback port), `--pass-env NAME`, `--pnpm-store DIR`, and a stderr line per refused connection (`nix/sandbox/sandbox.sh`, `nix/sandbox/egress-proxy.mjs`). Two capabilities the scorecard needs are missing: (a) visibility of allowed requests, either a log of each one or chaining its proxy to an upstream recording proxy, without which M27 can count only `CONNECT` tunnels and their hosts; (b) publishing a sandbox loopback port to host loopback, without which local rows cannot reach a running side. Both go to Kiro as requests to the launcher owner, and this plan never builds a second launcher. +- D2. **Starter deployment contract** (Lake 1 U13): the preview job records its URL and deployed sha as a GitHub Deployment (`environment: preview-pr-`, `environment_url`), and production does the same for `endgame.systemfsoftware.com`. The scorecard reads these over REST (KTD11). The request goes through Kiro to the Lake 1 builder. +- D3. **CI secrets from Kiro:** `CLOUDFLARE_URLSCANNER_API_TOKEN` (Account > URL Scanner, Edit) with `CLOUDFLARE_ACCOUNT_ID`, `ANTHROPIC_API_KEY` for M10, and a GitHub App token for pin-bump PRs, because PRs opened with `GITHUB_TOKEN` do not trigger `pull_request` workflows. +- D4. **Fleet runners** (`[self-hosted, systemfsoftware-runner, large]`) able to run the launcher. Lake 1 reports that this omp container's nix bubblewrap 0.12.0 fails `--proc /proc`, while this session's `/usr/sbin/bwrap` 0.11.2 ran `--unshare-all --proc /proc` with exit 0. U4 probes the fleet first. +- D5. **systemfsoftware flake packages** (systemfsoftware PR C, the same input Lake 1 U11b consumes) for the instrument's mutation leg: `@systemfsoftware/stryker-js` 15.0.1 and `@systemfsoftware/stryker-js-vitest-runner` 8.0.3, the versions the starter pins on Lake 1. Only U3's release-gate mutation job needs them. + +--- + +## Planning Contract + +### Key Technical Decisions + +- KTD1. **Beaten means the starter's worst run beats rat-stack's best.** A `measurement` row is `beaten` when every starter run is strictly better than every rat-stack run, so the two ranges do not overlap. A `count` row needs three identical runs on each side; any disagreement is an instrument error. Ties are `tie`, and a rat-stack value at a bound the starter cannot pass (0 bytes, level 5) stays `tie`. Range non-overlap over 3-5 runs is reproducibility a reviewer can check by eye, and it follows Lighthouse's own variability guidance (median of 5 is about twice as stable as 1 run, `GoogleChrome/lighthouse/docs/variability.md`). +- KTD2. **The noise band is `main`'s own run range.** A `measurement` regresses when the PR's median is worse than `main`'s worst run. A `count` regresses on any worse value. A row that goes from present to `absent` regresses. `no-secret` and fork-PR `absent` rows are neutral. The ratchet compares a row only when its metric-definition hash (the registry entry plus the files its family reads under `evals/ratstack-scorecard/`) matches `main`'s; a changed definition is reported as `re-baselined` and stays neutral until `main` carries it. A deliberate bin retirement is a verifier change to the registry, never a builder edit. This instantiates the gate Key Decision for R11 (session-settled: user-directed — chosen over a report-only job: Kiro's gate ruling, GATE1 sign-off 2026-10-05). +- KTD3. **The orchestrator is Deno with zero third-party imports, and everything else runs sandboxed.** `evals/ratstack-scorecard/src/**` imports only Deno APIs and `node:` builtins, and `deno.json` declares no import map. The decision modules (`*.workflow.ts`) import nothing at all, so Deno runs them and vitest tests them unchanged. Tools with npm code come from the instrument's own `package.json` and pnpm lockfile, are fetched through the lockfile-hash fixed-output derivation Lake 1 U11b adopts, and only ever run inside the launcher (R8). The tools are Lighthouse 13.5.0, oxc-parser 0.153.0, yaml 2.9.1, skills 1.7.0, plus, for tests, vitest 5.0.3, fast-check 4.10.2, `@fast-check/vitest` 0.5.0 and ajv 8.20.0. Deno and `node:` follow the repo's standalone-script rule without pulling JSR code that would itself need sandboxing. +- KTD4. **The instrument gets its own flake at `evals/ratstack-scorecard/flake.nix`.** It has its own `flake.lock`: nixpkgs at the root's current rev `4975466d324710c576dc11ad614684e6bd8cad8e`, plus the prm launcher input. It exposes `scorecard`, `ratstack-src`, `ratstack-toolchain` and `scorecard-tools`. The root `flake.nix` belongs to the Lake 1 builder, and a separate flake avoids co-ownership and keeps Dependabot from moving the instrument's nixpkgs. The instrument hash is the git tree hash of `evals/ratstack-scorecard/`. +- KTD5. **The pin is a file, bumped by the scorecard workflow.** `ratstack.pin.json` holds `{ repo, commit, narHash }`, and `ratstack-src` fetches it with a fixed-output fetch. A `pin` job runs on a daily schedule and on every push to `main`. It compares `git ls-remote` of `joelhooks/rat-stack` `main` with the pin and opens or updates one PR that changes only `ratstack.pin.json`, using the App token (D3), which is installed on this repository alone with `contents: write` and `pull-requests: write`. Kiro merges the bump like any other PR. The root Dependabot `nix` entry runs weekly and would double-bump a flake input, so the pin is not a flake input. +- KTD6. **Each side is measured through its documented entry points.** `src/sides/ratstack.ts` and `src/sides/starter.ts` hold, as data: install, gate, dev-stack start and readiness URL, ports, MCP endpoint, CLI entry, lockfile path, bins with their documented removal, the fake-secret env list, the vendored-path exclusions, and each command's egress allow-list. Provenance records the allow-list each run used. rat-stack's entries come from its README, AGENTS.md and CI (`pnpm install --frozen-lockfile`, `pnpm sources:fetch`, `pnpm turbo run check test build --concurrency=1`, `pnpm infra:dev` on `:1337`). The starter's come from its README and AGENTS.md (`pnpm check:ci`, `pnpm dev` / `bin/local-stack`). When an entry point drifts, the row reports `instrument-error`, and the verifier updates the adapter. +- KTD7. **Cold means nothing project-specific is cached on either side.** Each run uses a fresh clone, a fresh chroot Nix store (`--store` under that job's `RUNNER_TEMP`), an empty pnpm store and empty build caches, all deleted when the run ends, with toolchains from the scorecard flake (Node 24.20.0; pnpm 11.3.0 for rat-stack, fetched as a fixed-output tarball because rat-stack's `devEngines` fails on any other pnpm; the starter's own pinned pnpm). The sandbox's only egress is one mitmproxy 12.2.3 recording proxy, trusted through `NODE_EXTRA_CA_CERTS`, `SSL_CERT_FILE` and `NIX_SSL_CERT_FILE`, with the hosts the documented install needs on its allow-list (needs D1a). This makes the starter's Nix-fetched dependencies count against it exactly as rat-stack's registry fetches do. +- KTD8. **The debt counter is comment-aware and neutral.** It parses every tracked `*.{c,m,}{j,t}s{x,}` file with oxc-parser and counts comment directives in rat-stack's three families (`oxlint-disable*`, `@effect-diagnostics*`, `@ts-expect-error|ignore|nocheck`, rat-stack `scripts/oxlint-plugin-debt-ledger.ts:6-13`), plus `stryker-disable*`, `eslint-disable*`, `biome-ignore`, `dprint-ignore`, and `c8`/`istanbul`/`v8 ignore`. Exclusions are vendored trees only, each declared with its reason in the side adapter: rat-stack `tools/oxlint/anti-slop/`, `vendor/`; starter `repos/**`. Parsing comments rather than regex-matching bytes keeps a regex literal, such as the one inside rat-stack's own plugin, from counting as debt. The instrument never runs either side's counter, so the subject never produces the oracle (CONST-T10). +- KTD9. **Sabotages and removals are verifier-authored patches.** `sabotage//.patch` breaks one published behaviour present on both sides, such as Markdown-by-default negotiation, `/llms.txt`, MCP `tools/list`, or `/openapi.json`. M24's denominator is the intersection of behaviours both sides have. `removal//.patch` encodes the side's documented removal: rat-stack's README "Keep or cut" table and `skills/keep-or-cut/SKILL.md`, and the starter's R68 checklists. A patch that no longer applies is an `instrument-error` naming the patch. +- KTD10. **The agent row uses pinned Claude Code headless.** Claude Code 2.1.280 from nixpkgs runs in the launcher with `-p --bare --strict-mcp-config`, an `--mcp-config` naming only the side's HTTP MCP endpoint, `--max-turns 12`, a pinned `--model` id recorded in the metric definition, and `--json-schema` forcing `{ "answer": string }`. Its sandbox's project directory is an empty scratch directory, it has no built-in tools (`--tools ""`), so the side's MCP tools are its only tools, and `ANTHROPIC_API_KEY` reaches that sandbox alone through `--pass-env`. Questions live in `questions/agent-mcp.json`, and each oracle is computed from the side's lockfile (resolved `effect`, `typescript` and `alchemy` versions), so neither side's content authors the expected answer. A wrong answer or a timeout scores the cap (120 s). +- KTD11. **Networked subjects are found over REST.** The starter URL and served sha come from the GitHub Deployments API for the head sha (D2). The networked job finds the head sha's previews workflow run first, waits for it, and records `absent` immediately when none exists. rat-stack's served commit is the newest commit listed in `https://ratstack.sh/log.md`, the site's own build-time statement, compared with the pin for R6's flag. +- KTD12. **One workflow; families are data.** `.github/workflows/scorecard.yml` runs on `pull_request` to `main`, `push` to `main`, a daily `schedule` and `workflow_dispatch`. A `plan` job (`small`) prints the family list and cache keys from the registry. A `measure` matrix (`large`, one family per job, `fail-fast: false`) restores or produces each side's family JSON. Runs on `main` write the rat-stack cache that PR runs restore, because an Actions cache written on a PR ref is invisible to other PRs. An `aggregate` job (`small`) merges them, fetches the latest successful `main` artifact for the ratchet, writes the summary, uploads `scorecard.json` with `actions/upload-artifact@v7`, and fails per R11. Adding a family never changes the workflow. Lighthouse family jobs never share a machine with another Lighthouse run. +- KTD13. **Lighthouse is npm `lighthouse` 13.5.0 driving nixpkgs `chromium` 153.0.8010.52** (through `CHROME_PATH` and `--chrome-flags=--headless=new`) with default simulated throttling. nixpkgs' `lighthouse` attribute is sigp's Ethereum client, and `@lhci/cli` 0.15.1 pins Lighthouse 12.6.1, so neither is used. +- KTD14. **Readiness calls URL Scanner v2 with `agentReadiness: true`, then polls the result.** It submits `POST /accounts/{account_id}/urlscanner/v2/scan` with `agentReadiness: true` and polls `GET …/v2/result/{scan_id}` every 15 s. The first live call fixes where the option sits in the request (top level or `options`) and the result path. The cross-check posts `{ "url": … }` to `https://isitagentready.com/api/scan` and reads `level`, as rat-stack's `apps/mischief/scripts/smoke.sh` does. When the levels differ, the row is flagged. This implements the scanner Key Decision (session-settled: user-directed — chosen over isitagentready as primary: first-party reproducible JSON through our token). +- KTD15. **Tests are admitted by layer, with refusal as the default** (`skill://test-layer-selection`). Every decision lives in a `*.workflow.ts` module with a colocated `*.workflow.property.test.ts`, mutated at break 100 by sfs stryker-js in CI on push to `main` (D5), never locally. Family runners, harness, side adapters and REST clients are executors and adapters: they get no tests of their own, and each unit's family smoke run plus its PR sabotage proves them. Three process-isolated journeys under `evals/ratstack-scorecard/journeys/` observe what only the seam can see: J1 launcher isolation (U4), J2 the static family over a fixture tree through the real parser in the sandbox (U2), and J3 the CLI's exit status and schema-valid JSON on a ratchet failure (U3). Refused: tests of `registry.ts` (a declaration), unit tests of runners (they would spawn processes or mock the subject), and assertions on rendered Markdown wording. +- KTD16. **A mutated gate is graded only against a green baseline.** Every family that edits a clone and runs a gate (M13, M20, M22, M24, M25) first runs the unmodified gate on that side over warm caches, and a red baseline makes those cells `unmeasurable` with the first failing task. The cold path (M26) is its own baseline: a red cold gate makes M26 `unmeasurable` and leaves the other families alone. This keeps a kill rate or removal rate from being computed over a gate that was already failing. + +### High-Level Technical Design + +Run topology: the orchestrator is the only process outside the sandbox. It reads results and never imports subject code. + +```mermaid +flowchart TB + orch[Deno orchestrator, no third-party imports] -->|spawn| sbR[launcher: rat-stack at pin] + orch -->|spawn| sbS[launcher: starter at head] + orch -->|spawn| sbT[launcher: tools - Lighthouse, oxc, Claude Code, skills] + sbR -->|published port| orch + sbS -->|published port| orch + sbT -->|allow-listed loopback port| sbR + sbT -->|allow-listed loopback port| sbS + sbR -->|egress only via| proxy[mitmproxy recorder] + sbS -->|egress only via| proxy + orch -->|REST| ext[URL Scanner, isitagentready, ratstack.sh, GitHub Deployments] + orch --> json[family JSON per side] +``` + +CI flow and the ratchet: + +```mermaid +flowchart TB + trig[PR to main, push to main, daily, dispatch] --> plan[plan: families and cache keys] + plan --> measure[measure matrix, one family per job] + measure -->|rat-stack side| cache{cache hit on commit, instrument hash, nixpkgs rev?} + cache -->|yes| restore[restore rat-stack family JSON] + cache -->|no| runR[run rat-stack family, save cache] + measure -->|starter side| runS[run starter family] + restore --> agg[aggregate] + runR --> agg + runS --> agg + agg --> verdicts[verdict per row, KTD1] + verdicts --> ratchet{main artifact exists?} + ratchet -->|yes| cmp[compare with main, KTD2] + ratchet -->|no| first[record first baseline] + cmp --> out[summary table, scorecard.json, exit status per R11] + first --> out + trig -->|push to main, daily| pin[pin job: ls-remote rat-stack main] + pin -->|moved| pr[open or update the pin-bump PR] +``` + +Row status and verdict, as a directional sketch, not a specification: + +```text +CellStatus = Measured{runs[]} | Absent{reason} | Unsupported{citation} | Unmeasurable{error} + | NoSecret{name} | InstrumentError{error} +Verdict = match (ratstack, starter): + (Measured, Measured) -> compare per KTD1 -> Beaten | NotBeaten | Tie + (Unsupported, Measured) -> Beaten when the citation re-verified at the pin, else InstrumentError + (_, Absent | NoSecret) -> NotBeaten (neutral for the ratchet when NoSecret or fork-Absent) + (InstrumentError, _) | (_, InstrumentError) -> InstrumentError + (Unmeasurable, _) -> NotBeaten, surfaced for Kiro +``` + +### Output Structure + +```text +evals/ratstack-scorecard/ + README.md what each metric measures, how to run it, who owns it + flake.nix flake.lock KTD4 + deno.json no import map (KTD3) + package.json pnpm-lock.yaml pnpm-workspace.yaml instrument tools and test runner (KTD3) + vitest.config.ts stryker.config.ts KTD15 + scorecard.schema.json R10 data contract + ratstack.pin.json KTD5 + questions/agent-mcp.json KTD10 + sabotage/{ratstack,starter}/*.patch KTD9 + removal/{ratstack,starter}/*.patch KTD9 + src/ + main.ts CLI: plan, measure --family, aggregate, render, pin + model/cell.ts tagged unions (declaration) + model/*.workflow.ts every decision, zero imports + model/*.workflow.property.test.ts colocated properties (KTD15) + metrics/registry.ts + sides/{ratstack,starter}.ts + harness/{sandbox,proxy,stack,store}.ts + families/{static,cold-path,gate-mutation,running-stack,agent-surfaces,networked}.ts + tools/{count-directives,parse-lockfile,lighthouse-run}.mjs run only inside the launcher + journeys/ J1-J3, process-isolated, run inside the launcher (KTD15) +.github/workflows/scorecard.yml +``` + +### Assumptions + +These are the inferred bets the scoping confirmation would have covered. Kiro's plan approval confirms or redirects them. + +- The 29 metric families in the Metric Catalog are the per-bin metrics, extending Kiro's examples to every bin the Superiority Map names. +- Run counts are 3 for counts and cold clones, and 5 for Lighthouse, CLI latency and the agent row. +- rat-stack runs locally with throwaway values for every feature-enabling secret it declares, so its features are on (rat-stack `apps/mischief/src/worker.ts:96-166`). +- M11's starter analog is one hold-confirmation token redeemed 300 times. Until that capability exists, the starter cell is `absent`. +- M17 on rat-stack needs a local delivery endpoint. If rat-stack's delivery adapters (DROVR, POSTSHIBA) take a configurable base URL, the instrument serves a recording fake there. If they do not, the rat-stack cell is `unmeasurable` with the exact reason, and Kiro rules. +- The agent model id is the current Claude Sonnet id when U8 lands, recorded in the registry. Changing it changes the instrument hash and so re-baselines. + +### Sequencing + +U1 → U2 → U3 → U4 → U5 → U6 → U7 → U8 → U9. Each is one PR layer on the previous one. U3 wires CI once static rows exist. Every later unit only adds a family to the registry, so the published table grows one family per PR without the workflow changing. U2 onward waits on D1. U9's starter cells wait on D2. + +--- + +## Implementation Units + +### U1. Model the scorecard: cells, verdicts, ratchet, schema, rendering + +- **Goal:** The pure core decides every verdict and ratchet outcome from data and renders the table and JSON. +- **Requirements:** R2, R3, R4, R10, R11; KTD1, KTD2. +- **Dependencies:** D1 for running the tests (no unsandboxed mode); no unit dependency. +- **Files:** `evals/ratstack-scorecard/src/model/cell.ts`, `src/model/verdict.workflow.ts`, `src/model/ratchet.workflow.ts`, `src/model/render.workflow.ts`, `src/metrics/registry.ts` (M1-M29 definitions as data: id, bin, unit, direction, kind, runs, family, citations), `scorecard.schema.json`, `deno.json`, `package.json`, `pnpm-lock.yaml`, `pnpm-workspace.yaml`, `vitest.config.ts`, `stryker.config.ts`, `README.md`, `src/model/verdict.workflow.property.test.ts`, `src/model/ratchet.workflow.property.test.ts`, `src/model/render.workflow.property.test.ts`. +- **Approach:** + 1. Cells and verdicts are closed tagged unions, and every decision is one exhaustive dispatch with complexity 1 (CONST-D4, CONST-P2). + 2. `render.workflow.ts` emits Markdown under 1 MiB (the per-step summary cap), putting provenance in the JSON and only flags in the table. + 3. The schema has `schemaVersion: 1`, and every JSON the renderer emits validates against it. +- **Patterns to follow:** `repos/constitution/CONSTITUTION.md` Article I. The standalone-script shebang style from `scripts/check-changeset.ts`, with scoped `--allow-*` flags (OP15). +- **Test scenarios** (all properties over generated cells; the AEs are spec literals pinned beside them): + - AE1: overlapping ranges give `not-beaten`. Shifting the starter so its worst run is 409 gives `beaten`. + - AE2: unequal count runs give `instrument-error`. + - Swapping sides of a strictly separated `measurement` turns `beaten` into `not-beaten`, and identical ranges are always `tie`. + - Direction `higher` mirrors direction `lower` under negation. + - AE3: an `Unsupported` cell whose citation check failed gives `instrument-error`, and one that passed gives `beaten`. + - AE5: a PR median inside `main`'s run range passes the ratchet, and a median just past `main`'s worst run fails, naming the metric. + - AE4: `beaten` on `main` and `not-beaten` on the PR fails the ratchet, and the failure names the metric and the cell status that caused it (for example a rat-stack `Unmeasurable` cell, or a changed live commit). + - Present on `main` and `absent` on the PR fails, while `NoSecret`, fork `Absent`, or a changed metric-definition hash (`re-baselined`) is neutral. + - With no `main` artifact, the ratchet records a first baseline and passes. + - Every rendered JSON validates against `scorecard.schema.json` (ajv), and the hand-written refusal holds: a document without `provenance.commit` fails validation. +- **Verification:** `pnpm vitest run` from `evals/ratstack-scorecard` under the launcher is green. One sabotage, flipping `<` to `<=` in the KTD1 comparison, turns a property red. + +### U2. Pin rat-stack and measure the static rows + +- **Goal:** The pin, the instrument flake and the static family produce M23 and M28 for both sides, plus the citation check that M12 relies on. +- **Requirements:** R1, R3, R4, R5, R8; KTD3, KTD4, KTD5, KTD8. +- **Dependencies:** U1, D1. +- **Files:** `evals/ratstack-scorecard/flake.nix`, `flake.lock`, `ratstack.pin.json` (commit `54d356037c994f89698760a4727be71d0005a087`), `src/sides/ratstack.ts`, `src/sides/starter.ts`, `src/harness/sandbox.ts`, `src/families/static.ts`, `src/tools/count-directives.mjs`, `src/tools/parse-lockfile.mjs`, `src/main.ts`, `src/model/directives.workflow.ts`, `src/model/citation.workflow.ts`, `src/model/directives.workflow.property.test.ts`, `src/model/citation.workflow.property.test.ts`, `journeys/static.journey.test.ts` (J2), `journeys/__fixtures__/debt-tree/`. +- **Approach:** + 1. `ratstack-src` fetches the pin as a fixed-output derivation. + 2. `scorecard-tools` builds the tools' `node_modules` through the same lockfile-hash fixed-output derivation Lake 1 U11b adopts. + 3. `sandbox.ts` wraps the launcher. Both tools run inside it read-only over the checkout. + 4. `parse-lockfile.mjs` reads the `packages` keys in both pnpm 11 and pnpm 12 lockfile formats. +- **Patterns to follow:** `bin/dprint` for the nix-run wrapper idea. rat-stack `apps/mischief/scripts/content-lib.ts:866-895` for which files count as source, adapted per KTD8. +- **Test scenarios:** + - Property (`directives.workflow`): a comment text carrying any directive from the KTD8 families classifies to that family, and a generated text with none classifies to nothing. + - Property (`directives.workflow`): a path under a declared vendored root is excluded and carries its reason, whatever its extension. + - Property (`citation.workflow`): a citation verifies only when the cited line range of the given file bytes contains the declared text. Shifting the text one line outside the range fails it. + - J2: a fixture tree with a directive in a comment, the same text in a string, a regex literal, and a vendored path counts only the comment, through the real oxc-parser in the launcher. Its pnpm 11 and pnpm 12 lockfile fixtures yield the same distinct `name@version` set through the real yaml parser. +- **Verification:** + - `nix run ./evals/ratstack-scorecard#scorecard -- measure --family static` prints M23 and M28 for both sides, identical across three runs. + - The PR body records M23 for rat-stack at the pin beside its published `/debt.md` total (217 on 2026-10-05), with the extra KTD8 families broken out. + - Sabotage evidence: invoking the counter with the launcher wrapper removed is refused, because `sandbox.ts` has no unsandboxed path. + +### U3. Publish the table in CI with the ratchet and the pin-bump job + +- **Goal:** Every PR to `main` and every push to `main` publishes the table and JSON. The job fails per R11, and a moved rat-stack `main` opens a pin-bump PR. +- **Requirements:** R7, R10, R11, R12; KTD2, KTD5, KTD12. +- **Dependencies:** U2, D3 (App token), D4, D5. +- **Files:** `.github/workflows/scorecard.yml`, `evals/ratstack-scorecard/src/main.ts` (`plan`, `aggregate`, `pin` commands), `src/harness/store.ts` (`main` artifact fetch), `src/model/pin.workflow.ts`, `src/model/cache-key.workflow.ts`, `src/model/pin.workflow.property.test.ts`, `src/model/cache-key.workflow.property.test.ts`, `journeys/aggregate.journey.test.ts` (J3), `journeys/__fixtures__/families/`. +- **Approach:** + 1. Permissions are `contents: read` and `actions: read`; only the `pin` job adds `contents: write` and `pull-requests: write` through the App token. + 2. Actions are tag-pinned as the repo does: `actions/checkout@v7`, `actions/cache@v6`, `actions/upload-artifact@v7`, `cachix/install-nix-action@v31`. + 3. Fleet labels follow Lake 1 KTD2. Concurrency is `scorecard-`, cancelling superseded PR runs and never `main` runs. + 4. A `mutation` job on push to `main` runs sfs stryker-js over `src/model/*.workflow.ts` at break 100 (KTD15, D5). It never runs on PRs or locally. + 5. The PR body declares the Evaluator change and cites Kiro's 2026-10-05 gate ruling as its GATE1 approval. +- **Patterns to follow:** Lake 1 `.github/workflows/ci.yml` and `release-gate.yml` (`plan` job then matrix, `upload-artifact`). +- **Test scenarios:** + - Property (`pin.workflow`): an `ls-remote` commit equal to the pin plans nothing, and any other commit plans a bump carrying that commit and the prefetched `narHash`. + - Property (`cache-key.workflow`): the rat-stack key changes when the instrument tree hash, the rat-stack commit or the nixpkgs rev changes, and stays the same when only the starter commit changes. + - J3: `main.ts aggregate` over fixture family JSONs where `main` beats M14 and the PR does not exits non-zero, names M14, and writes a `scorecard.json` that validates against the schema (AE4). +- **Verification:** The PR's own scorecard run publishes M23 and M28, with the starter side being the `hello` seed. actionlint is clean. Sabotage evidence: a registry entry with unequal count runs turns the job red, and reverting it turns the job green. + +### U4. Run both sides in the sandbox launcher + +- **Goal:** The harness starts, probes, and stops each side's documented local stack inside the launcher, with published ports, fake secrets and the recording proxy. +- **Requirements:** R5, R8; KTD6, KTD7; AE8. +- **Dependencies:** U2, D1, D4. +- **Files:** `evals/ratstack-scorecard/src/harness/stack.ts`, `src/harness/proxy.ts`, `src/sides/ratstack.ts`, `src/sides/starter.ts`, `flake.nix` (`ratstack-toolchain`: Node 24.20.0, pnpm 11.3.0 tarball), `journeys/sandbox.journey.test.ts` (J1). +- **Approach:** + 1. Probe the fleet first: `--unshare-all` plus whatever `/proc` mode the launcher uses. A failure goes to Kiro with the exact error (D4). + 2. Readiness: rat-stack `:1337/llms.txt` under `pnpm infra:dev`; the starter's readiness URL from its adapter. An `absent` starter stack gives `absent` rows. +- **Test scenarios:** + - J1 (AE8): from inside the rat-stack sandbox, reading `~/.ssh` fails, writing outside the project fails, and a connection to an undeclared host fails. +- **Verification:** + - Both stacks reach ready on the fleet, and the J1 probe results appear in the run's provenance. + - A request to an allow-listed host appears in the proxy log with its host. + - Sabotage evidence: running the J1 probes without the launcher succeeds, which turns J1 red. + +### U5. Measure the cold path: clone to green, install requests, build sizes + +- **Goal:** M26, M27, M6 and M7 for both sides under KTD7. +- **Requirements:** R1, R3; KTD7. +- **Dependencies:** U4. +- **Files:** `evals/ratstack-scorecard/src/families/cold-path.ts`, `src/model/cold.workflow.ts`, `src/model/cold.workflow.property.test.ts`. +- **Approach:** + 1. One cold run yields all four metrics: time from clone to green gate, proxy request count with hosts, and gzipped bytes of the side's declared client JS and Worker outputs (rat-stack `apps/mischief/dist/startup/worker.bundle`, `apps/web` build JS; starter outputs from its adapter). + 2. rat-stack's gate is its CI triple, including `pnpm sources:fetch`, whose `github.com` egress is allow-listed. A red unmodified gate makes M26 `unmeasurable` (KTD16). +- **Test scenarios** (`cold.workflow` properties): + - A store listing that contains any path from the side's declared dependency closure is judged not cold, and the run is refused. + - A gate outcome with a non-zero exit becomes `unmeasurable` carrying the failing step, never a time. + - Any generated proxy log reduces to a request count equal to its entry count and a host set equal to its distinct hosts. +- **Verification:** Three cold runs per side complete on `large`, and the times and request counts land in the family JSON. + +### U6. Grade gates by breaking things: sabotage, removal, quotes, bindings, adapters + +- **Goal:** M13, M20, M21, M22, M24 and M25 for both sides. +- **Requirements:** R1, R4; KTD9. +- **Dependencies:** U5. +- **Files:** `evals/ratstack-scorecard/src/families/gate-mutation.ts`, `src/model/gate-mutation.workflow.ts`, `src/model/gate-mutation.workflow.property.test.ts`, `sabotage/ratstack/*.patch`, `sabotage/starter/*.patch`, `removal/ratstack/*.patch` (CLI-only, no code mode, no HTTP, no MCP, no analytics, no XState, no devtools, per rat-stack README "Keep or cut"), `removal/starter/*.patch`. +- **Approach:** + 1. Each side first runs its unmodified gate (KTD16). Each case then runs in a throwaway clone from U5's cold tree (warm caches allowed; timing is not measured here). + 2. M21 imports the stack module inside a sandbox with no egress and counts refused connections from the launcher log. + 3. M13 flips the side's documented adapter switch (rat-stack `DatabaseVendor`). + 4. Cases with no counterpart on a side record `absent` or `unsupported` per R4. +- **Test scenarios** (`gate-mutation.workflow` properties): + - A patch that fails to apply becomes `instrument-error` naming the patch. + - A red baseline makes every case on that side `unmeasurable`, whatever the case outcomes. + - Over a green baseline, a case whose gate stays green is survived (M24) or kept (M25), and a red one is killed (M24) or broken with its first failing task (M25). + - M24's denominator is the intersection of behaviours both sides declare, so adding a behaviour on one side never changes the other side's rate. +- **Verification:** All rat-stack cases run once and are cached under the R12 key, and starter cases on `main` report `absent` where the bin does not exist yet. + +### U7. Race and break the running stacks + +- **Goal:** M11, M12, M14, M17 and M19 for both sides. +- **Requirements:** R1, R4; Key Decision on race rows. +- **Dependencies:** U4. +- **Files:** `evals/ratstack-scorecard/src/families/running-stack.ts`, `src/model/running-stack.workflow.ts`, `src/model/running-stack.workflow.property.test.ts`. +- **Approach:** + 1. M11: on rat-stack, mint one page ticket from `/tokenmaxx` agent Markdown and send 300 concurrent `POST /api/joinInterest` with distinct submissions (rat-stack `apps/mischief/src/interest/page-ticket.ts:12-35`, `packages/intake-live/src/hmac-tickets.ts:139-163`). + 2. M12: the starter's documented claim capability; rat-stack is `Unsupported` with the cited text re-verified at the pin. + 3. M14: probe both sides' auth with email + password sign-up and sign-in on their documented auth base paths. + 4. M17: SIGKILL the side's runtime process at a random point in a 50-request burst, then tear down that sandbox (its PID namespace dies with it) and start a fresh one over the same project state, drain, and count messages per request in the mail sink. + 5. M19: overshoot is measured against the bound read from the side's own configuration. +- **Test scenarios** (`running-stack.workflow` properties): + - For any response set, M11 equals the number of successful redemptions minus one, floored at zero. + - M12 equals seats granted minus capacity, floored at zero. + - M14 counts a 2xx as one and a 4xx as zero, and any 5xx makes the cell `unmeasurable`. + - M17 counts a request with 0 or 2+ sink messages, never one with exactly 1. + - M19 equals accepted requests minus the side's declared bound, floored at zero. +- **Verification:** + - Three runs agree on every count for rat-stack at the pin, and the results are cached. + - Sabotage evidence in the PR body: M11 pointed at a stub endpoint that accepts every redemption reports 299, which proves the race can see oversell. + +### U8. Measure the agent surfaces + +- **Goal:** M4, M5, M8, M9, M10, M15, M16, M18 and M29 for both sides. +- **Requirements:** R1, R8; KTD10. +- **Dependencies:** U4, D3 (`ANTHROPIC_API_KEY`). +- **Files:** `evals/ratstack-scorecard/src/families/agent-surfaces.ts`, `src/model/agent-surfaces.workflow.ts`, `src/model/agent-surfaces.workflow.property.test.ts`, `questions/agent-mcp.json`. +- **Approach:** + 1. Surfaces are driven over HTTP, MCP 2026-07-28 (with the `Mcp-Method` and `Mcp-Name` headers rat-stack's smoke uses), CLI, RPC, code mode, A2A and gRPC where the side serves them. + 2. M8's inputs come from each capability's OpenAPI examples. A capability without an example scores 0 surfaces. + 3. M10 runs Claude Code in its own sandbox with egress to `api.anthropic.com` and the side's published MCP port only. + 4. M15 crawls the side's own `/sitemap.xml`. + 5. M29 runs `skills add` in a sandbox per skill. +- **Test scenarios** (`agent-surfaces.workflow` properties): + - An M10 answer scores its elapsed seconds only on exact string equality with the lockfile oracle. Any other answer, or a timeout, scores 120 s. + - A missing `ANTHROPIC_API_KEY` yields `NoSecret` for M10 on that side. + - M4 counts exactly the served versions that differ from the lockfile resolution, so `effect 4.0.0` served against `4.0.1` locked counts one. + - M8 counts a surface only when its normalized output deep-equals HTTP's, and normalization is idempotent. + - M5 is the share of entries with both a PR link and a CI-run link, in the range 0 to 1. +- **Verification:** rat-stack rows are measured at the pin and cached, and the starter's are `absent` until its surfaces exist. + +### U9. Measure the networked rows against live sites + +- **Goal:** M1, M2 and M3 against live `ratstack.sh` and the starter's deployment, with live-commit flags. +- **Requirements:** R6; KTD11, KTD13, KTD14; AE6, AE7. +- **Dependencies:** U3, D2, D3 (URL Scanner token). +- **Files:** `evals/ratstack-scorecard/src/families/networked.ts`, `src/tools/lighthouse-run.mjs`, `src/model/networked.workflow.ts`, `src/model/networked.workflow.property.test.ts`. +- **Approach:** + 1. Lighthouse runs five times serially inside the launcher, with egress limited to the target host. + 2. The rat-stack networked cache key uses the live served commit in place of the pin. + 3. URL Scanner calls respect its 1-per-10 s limit. +- **Test scenarios** (`networked.workflow` properties): + - AE6: a `log.md` text whose newest listed commit differs from the pin sets the `live≠pin` flag, and an equal commit clears it. + - AE7: an empty previews-run list for the head sha yields `Absent` immediately, with no wait. + - Different URL Scanner and isitagentready levels set the disagreement flag, and M1 keeps the URL Scanner value. + - Decoding a URL Scanner result without the agent-readiness payload yields `instrument-error` quoting the keys it received. The first live response is saved as `src/model/__fixtures__/url-scanner-result.json`, the Cloudflare-authored oracle the decode property runs over together with generated key deletions. +- **Verification:** The first live run against `ratstack.sh` records level, checks and Lighthouse values. The starter cells read `absent` until Lake 1 U13 deploys. + +--- + +## Verification Contract + +| Gate | Command | Applies to | +| --------------------- | ------------------------------------------------------------------------------------- | ---------------------------------------------- | +| Format (START-1) | `pnpm format:check` (dprint covers `evals/**`) | every unit | +| Instrument properties | `pnpm vitest run --project model` under the launcher, from `evals/ratstack-scorecard` | every unit | +| Journeys J1-J3 | `pnpm vitest run --project journeys` under the launcher | U2 (J2), U3 (J3), U4 (J1) and every later unit | +| Family run | `nix run ./evals/ratstack-scorecard#scorecard -- measure --family ` | U2, U5-U9 | +| Workflow lint | `actionlint` | U3 | +| Repo gate (START-4) | `pnpm check:ci` | every unit before its PR | +| Sabotage | break one decision or one probe, show the test red, revert | every unit; recorded in the PR body | + +Mutation testing never runs locally; it runs only in the scorecard workflow's `mutation` job on push to `main` (KTD15). PR bodies carry commands, outputs and sabotage evidence, and the verifier runs every check in this session (VER1). + +## Definition of Done + +- U1-U9 are merged to `main` as one stack, each PR green on its own gate with sabotage evidence in its body. +- On `main`, the scorecard run publishes all 29 families. Every rat-stack cell is `Measured`, `Unsupported` with a verified citation, or `Unmeasurable` with an exact error that Kiro has ruled on. No row is missing. +- A PR to `main` shows the summary table and a `scorecard.json` artifact that validates against the schema, and AE4's sabotage turns that PR's scorecard job red. +- The pin-bump job has opened at least one PR, or shows "pin current" against `git ls-remote`. +- Each PR appended its surprise and one proposed AGENTS.md line to `docs/brainstorms/REFLECTION.md`, including the R9 ownership line. +- No experimental or abandoned code remains under `evals/ratstack-scorecard/`. + +--- + +## Risks + +| Risk | Mitigation | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Some rows sit at a bound the starter cannot pass. rat-stack's home page ships no application scripts (rat-stack `.brain/projects/ratstack-sh/plain-svelte-content-compiler.svx:93`) and it claims readiness level 5, so a TanStack home page may lose or tie M3 and M6. | The instrument reports `tie` or `not-beaten` truthfully (KTD1). Kiro sees which Superiority Map claims the numbers contradict. The instrument is never tuned to make them pass. | +| The ratchet turns PRs red when live `ratstack.sh` deploys an improvement. | This is intended ("beat the current rat-stack"). The cache key includes the live commit, so a flip happens once per rat-stack deploy, and the table names the new live commit. | +| The undocumented isitagentready API changes shape. | It is a cross-check row only. A shape change is an `instrument-error` quoting the keys it received. | +| The launcher lacks port publishing or visibility of allowed requests (D1a, D1b). The scorecard runs on the Linux fleet only, so the launcher's macOS profile is not on its path. | D1 lists the exact capabilities. The gap goes to Kiro for the launcher owner. No parallel launcher. | +| Fleet hardware varies between the rat-stack baseline and starter runs. | Provenance records the runner name. Timing rows need non-overlapping ranges (KTD1). A re-baseline is one instrument-hash change away. | +| Verifier-authored patches rot as the starter changes. | A non-applying patch is a loud `instrument-error` (KTD9) that the verifier fixes. Builders never touch it (R9). | +| Per-PR runtime. The starter side runs every family on every PR (Key Decisions), including three cold clones and one gate run per sabotage and per removal. | Families run in parallel on separate `large` runners. Each family job has a `timeout-minutes` ceiling recorded in the registry, and a timeout is an `instrument-error` naming the family, never a silent skip. Provenance records each family's wall time so Kiro can re-rule on cost with numbers. | + +## Sources + +- Origin: `docs/brainstorms/inputs/requirements-final.md` (Superiority Map; R46, R51, R68, R69), `docs/brainstorms/inputs/ruling-nix-distribution-sandbox.md`, `docs/brainstorms/inputs/2026-10-05-starter-ratstack-brief.md`. +- Lake 1 plan (sibling worktree `brainstorm`): `docs/plans/2026-10-05-2014-feat-starter-lake-1-foundation-plan.md`, KTD2 (fleet labels), KTD14 (local stack), U11b (sandbox), U13 (previews), Kiro Rulings item 6 (bubblewrap probe). +- rat-stack at `54d3560`: `package.json` (pnpm 11.3.0, `devEngines`), `.github/workflows/ci.yml:18-23`, `scripts/acceptance-cold-clone.sh`, `scripts/oxlint-plugin-debt-ledger.ts:6-13`, `apps/mischief/scripts/content-lib.ts:866-895`, `apps/mischief/src/app.ts:510-556`, `apps/mischief/src/worker.ts:96-166,363`, `apps/mischief/scripts/smoke.sh`, `packages/core/src/join-interest-contract.ts:22-26`, `apps/mischief/src/interest/interest-durable-object.ts:60-75`, README "Keep or cut". +- Live: `https://ratstack.sh/debt.md` (217 directives), `https://ratstack.sh/log.md` (newest `ed63ba3`, 2026-10-03); `git ls-remote https://github.com/joelhooks/rat-stack` HEAD `54d356037c994f89698760a4727be71d0005a087`. +- Versions verified 2026-10-05: npm `lighthouse` 13.5.0, `oxc-parser` 0.153.0, `yaml` 2.9.1, `fast-check` 4.10.2, `skills` 1.7.0, `pnpm` 11.3.0; nixpkgs `4975466d` `chromium` 153.0.8010.52, `claude-code` 2.1.280, `mitmproxy` 12.2.3, `nodejs_24` 24.20.0, `deno` 2.9.6, `bubblewrap` 0.12.0; GitHub `actions/upload-artifact` v7.0.1, `actions/checkout` v7.0.1. +- Cloudflare URL Scanner: `developers.cloudflare.com/api/resources/url_scanner/subresources/scans/methods/create/`, scan limits `developers.cloudflare.com/security-center/investigate/scan-limits/`, Agent Readiness levels `blog.cloudflare.com/agent-readiness/`. Lighthouse variability: `github.com/GoogleChrome/lighthouse/blob/main/docs/variability.md`. Job summary limit: `docs.github.com/en/actions/reference/workflows-and-actions/workflow-commands`. Claude Code headless flags: `code.claude.com/docs/en/cli-reference`. From f9f2fa3db681b0b7869d5f50ac592922dc6b0fd7 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Mon, 5 Oct 2026 22:26:13 +0000 Subject: [PATCH 2/7] feat(repo): model scorecard verdicts, the ratchet and the published schema The scorecard's pure core. judgeRow decides beaten, not beaten, tie or instrument error for one row: the starter's worst run must be strictly better than rat-stack's best (KTD1), a count must reproduce exactly, and an unsupported rat-stack invariant counts only while its citation holds. compareWithMain is the ratchet (KTD2): it fails on instrument errors, on a row beaten on main and lost on the PR, and on a starter value worse than main's worst run, and treats missing secrets, fork PRs and changed metric definitions as neutral. scorecard.schema.json is the JSON contract CI publishes and /scorecard will render. registry.ts declares the M1-M29 rows as data. The decision modules import only each other, so Deno checks them with no third-party code. Their laws run under vitest inside the sandbox launcher from pnpm-release-management --- evals/ratstack-scorecard/.gitignore | 1 + evals/ratstack-scorecard/README.md | 31 + evals/ratstack-scorecard/deno.json | 9 + evals/ratstack-scorecard/package.json | 11 + evals/ratstack-scorecard/pnpm-lock.yaml | 692 ++++++++++++++++++ evals/ratstack-scorecard/pnpm-workspace.yaml | 3 + .../ratstack-scorecard/scorecard.schema.json | 303 ++++++++ .../src/metrics/registry.ts | 377 ++++++++++ .../src/model/acceptance-examples.test.ts | 104 +++ evals/ratstack-scorecard/src/model/cell.ts | 125 ++++ ...ompare-with-main.workflow.property.test.ts | 101 +++ .../src/model/compare-with-main.workflow.ts | 110 +++ .../ratstack-scorecard/src/model/dispatch.ts | 38 + .../model/judge-row.workflow.property.test.ts | 92 +++ .../src/model/judge-row.workflow.ts | 89 +++ evals/ratstack-scorecard/src/model/runs.ts | 12 + .../model/scorecard-document.property.test.ts | 34 + .../src/model/scorecard-document.ts | 78 ++ .../src/model/scorecard.arbitrary.ts | 139 ++++ .../src/model/summary-table.property.test.ts | 20 + .../src/model/summary-table.ts | 100 +++ evals/ratstack-scorecard/vitest.config.ts | 10 + 22 files changed, 2479 insertions(+) create mode 100644 evals/ratstack-scorecard/.gitignore create mode 100644 evals/ratstack-scorecard/README.md create mode 100644 evals/ratstack-scorecard/deno.json create mode 100644 evals/ratstack-scorecard/package.json create mode 100644 evals/ratstack-scorecard/pnpm-lock.yaml create mode 100644 evals/ratstack-scorecard/pnpm-workspace.yaml create mode 100644 evals/ratstack-scorecard/scorecard.schema.json create mode 100644 evals/ratstack-scorecard/src/metrics/registry.ts create mode 100644 evals/ratstack-scorecard/src/model/acceptance-examples.test.ts create mode 100644 evals/ratstack-scorecard/src/model/cell.ts create mode 100644 evals/ratstack-scorecard/src/model/compare-with-main.workflow.property.test.ts create mode 100644 evals/ratstack-scorecard/src/model/compare-with-main.workflow.ts create mode 100644 evals/ratstack-scorecard/src/model/dispatch.ts create mode 100644 evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts create mode 100644 evals/ratstack-scorecard/src/model/judge-row.workflow.ts create mode 100644 evals/ratstack-scorecard/src/model/runs.ts create mode 100644 evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts create mode 100644 evals/ratstack-scorecard/src/model/scorecard-document.ts create mode 100644 evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts create mode 100644 evals/ratstack-scorecard/src/model/summary-table.property.test.ts create mode 100644 evals/ratstack-scorecard/src/model/summary-table.ts create mode 100644 evals/ratstack-scorecard/vitest.config.ts diff --git a/evals/ratstack-scorecard/.gitignore b/evals/ratstack-scorecard/.gitignore new file mode 100644 index 0000000..ceddaa3 --- /dev/null +++ b/evals/ratstack-scorecard/.gitignore @@ -0,0 +1 @@ +.cache/ diff --git a/evals/ratstack-scorecard/README.md b/evals/ratstack-scorecard/README.md new file mode 100644 index 0000000..7c75285 --- /dev/null +++ b/evals/ratstack-scorecard/README.md @@ -0,0 +1,31 @@ +# rat-stack scorecard + +The scorecard measures the starter against [rat-stack](https://github.com/joelhooks/rat-stack) on one or more metrics per rat-stack bin and surface, and publishes which bins the starter beats. Both sides are run; no number comes from reading a README. + +## Ownership + +`evals/ratstack-scorecard/**` and `.github/workflows/scorecard.yml` are an Evaluator surface owned by the verifier session (`starter-verify`). Builders never edit them: a builder who changes the instrument that grades their work reports the score they chose (CONST-E9). A metric that looks wrong goes to Kiro as a finding. + +## What it measures + +`src/metrics/registry.ts` lists every row: its id (`M1`-`M29`, with sub-rows such as `M3.lcp`), the rat-stack bin it grades, its unit, which direction is better, whether it is a `count` (three runs that must agree exactly) or a `measurement` (N runs), and the family that measures it. The plan's Metric Catalog (`docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md`) explains each row. + +## How a verdict is reached + +- A row is **beaten** only when the starter's worst run is strictly better than rat-stack's best run. Overlapping ranges are **not beaten**; identical ranges are a **tie**. +- A count that does not reproduce exactly, or a side that produced the wrong number of runs, is an **instrument error**. +- An invariant rat-stack lacks is **unsupported** with a `file:line` citation. The citation is re-checked at the pin; if the cited lines no longer say it, the row is an instrument error. +- The ratchet compares a PR with the latest `main` artifact. It fails on any instrument error, on a row beaten on `main` that the PR no longer beats, and on a starter value worse than `main`'s worst run. A missing secret or a fork PR without a preview is neutral. A changed metric definition is re-baselined. + +## Running it + +Everything that loads third-party code runs inside the sandbox launcher from `systemfsoftware/pnpm-release-management` (`packages..sandbox`), with this directory as the sandbox project: + +```sh +cd evals/ratstack-scorecard +export SANDBOX_PROJECT=$PWD +sandbox --allow-host registry.npmjs.org -- pnpm install --frozen-lockfile +sandbox -- pnpm vitest run --project model +``` + +The decision modules (`src/model/*.workflow.ts`) and the orchestrator import only Deno APIs, `node:` builtins and each other, so `deno check src/` type-checks them without any third-party code. diff --git a/evals/ratstack-scorecard/deno.json b/evals/ratstack-scorecard/deno.json new file mode 100644 index 0000000..0b79998 --- /dev/null +++ b/evals/ratstack-scorecard/deno.json @@ -0,0 +1,9 @@ +{ + "compilerOptions": { + "strict": true, + "exactOptionalPropertyTypes": true, + "noUncheckedIndexedAccess": true + }, + "exclude": ["node_modules/", "**/*.test.ts", "**/*.arbitrary.ts", "vitest.config.ts"], + "lock": false +} diff --git a/evals/ratstack-scorecard/package.json b/evals/ratstack-scorecard/package.json new file mode 100644 index 0000000..7f59206 --- /dev/null +++ b/evals/ratstack-scorecard/package.json @@ -0,0 +1,11 @@ +{ + "name": "@starter/ratstack-scorecard", + "private": true, + "type": "module", + "devDependencies": { + "@fast-check/vitest": "0.5.0", + "ajv": "8.20.0", + "fast-check": "4.10.2", + "vitest": "5.0.3" + } +} diff --git a/evals/ratstack-scorecard/pnpm-lock.yaml b/evals/ratstack-scorecard/pnpm-lock.yaml new file mode 100644 index 0000000..05f1fa0 --- /dev/null +++ b/evals/ratstack-scorecard/pnpm-lock.yaml @@ -0,0 +1,692 @@ +lockfileVersion: '9.0' + +settings: + autoInstallPeers: true + excludeLinksFromLockfile: false + +importers: + + .: + devDependencies: + '@fast-check/vitest': + specifier: 0.5.0 + version: 0.5.0(vitest@5.0.3(vite@8.3.2)) + ajv: + specifier: 8.20.0 + version: 8.20.0 + fast-check: + specifier: 4.10.2 + version: 4.10.2 + vitest: + specifier: 5.0.3 + version: 5.0.3(vite@8.3.2) + +packages: + + '@fast-check/vitest@0.5.0': + resolution: {integrity: sha512-Kgbj2smuq1P4OhGKuPATs5i2fAxNmGbJ/zSmTrzWhQSLUrHREbKWgv9WGbNUOXcx77q6joLVMmTygyXuwP2Bog==} + peerDependencies: + vitest: ^4.1.0 || ^5.0.0 + + '@jridgewell/resolve-uri@3.1.2': + resolution: {integrity: sha512-bRISgCIjP20/tbWSPWMEi54QVPRZExkuD9lJL+UIxUKtwVJA8wW1Trb1jMs1RFXo1CBTNZ/5hpC9QvmKWdopKw==} + engines: {node: '>=6.0.0'} + + '@jridgewell/sourcemap-codec@1.6.0': + resolution: {integrity: sha512-T7jf+5zgsZHwNJ4lvQ7/aezbyk0nNX+zJVWpmHA7VYsEx7a7qr5Rg5IbtJFqkgze5Y2sruq1RUY8Q837Od7iFw==} + + '@jridgewell/trace-mapping@0.3.31': + resolution: {integrity: sha512-zzNR+SdQSDJzc8joaeP8QQoCQr8NuYx2dIIytl1QeBEZHJ9uW6hebsrYgbz8hJwUQao3TWCMtmfV8Nu1twOLAw==} + + '@oxc-project/types@0.152.0': + resolution: {integrity: sha512-oM/5rLBm2tPkg0iBgkH/FOeR3PCDpY19GTgAZjMFM8h9WI9VW7cLgzp6nwtarYKmovavIQZ+Fe/RKX/8C8O/Rw==} + + '@rolldown/binding-android-arm-eabi@1.2.12': + resolution: {integrity: sha512-dB/a1214qKfHMXCpgqR4OZT+jS4kTyEXbQGJPqzobt5EwH5rX080pxE37alt3RzvR1bf1Yz/yGqRfrYAxuPw0A==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm] + os: [android] + + '@rolldown/binding-android-arm64@1.2.12': + resolution: {integrity: sha512-7KHFgQ5VJxIHcLlrwrc3Xbds7oTNQT7Pgi9gQCJKrd2VGab/UksIOYp6VD8MzCstGxOKMgNamPwUCfxPdP1OHg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [android] + + '@rolldown/binding-darwin-arm64@1.2.12': + resolution: {integrity: sha512-3YIhqHD96nA5SaYNRBR16HnGv4oavZvXfD/ayHM+oYZ0WD/8lBAtf6zQua4kEyAvpqrluKXl0lnOBoiNby7x9w==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [darwin] + + '@rolldown/binding-darwin-x64@1.2.12': + resolution: {integrity: sha512-UuuJ35MFw4gmFOrE9pEqIV+K3syIKveph+Qc1/ljHZVdoDW4pz/JHR/eMVom+TZGl/5OOvGJOWaOCVt3ZfqhxA==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [darwin] + + '@rolldown/binding-freebsd-x64@1.2.12': + resolution: {integrity: sha512-uMvssit0a4W+/7D8CbHUvG719mH3R2jwXAlh/XcPvuHTE0g++LymF88DCGNX0HM2rBOn0xrzgXktIB6fLSJBTQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [freebsd] + + '@rolldown/binding-linux-arm-gnueabihf@1.2.12': + resolution: {integrity: sha512-XcFu0R0xWnwzSf4IQgFH1rJIckPN1pLy2R+4r9IDB7Yfu/ys9cVqfa4pBrMHj7a3gl8mIR4nRNPg0e5IvEVs6g==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm] + os: [linux] + + '@rolldown/binding-linux-arm64-gnu@1.2.12': + resolution: {integrity: sha512-260UrKgn8tz39ak+SMDOirKzr7V04M9dWPw5llW00SwBivCZoWcRBKV1d8cXnRkUmSZA3BdiUmBHWk7734Ulpw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [linux] + libc: [glibc] + + '@rolldown/binding-linux-arm64-musl@1.2.12': + resolution: {integrity: sha512-5YK1I9SqDkbPgc1IA8BgDl34suqUS2q0KWnBrirm0E51YjOs6eo6dV6jbQfNE/argHRSvd0QUGgtpIoYx+WWpw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [linux] + libc: [musl] + + '@rolldown/binding-linux-ppc64-gnu@1.2.12': + resolution: {integrity: sha512-Rkcrmp7eFRg74yL5fXEU91JEWbdEPLevWwGtXpmhbjlD1StScbWTmO94Bhly+Mo+ketKYkdmM1vNUKeWSlx8cQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [ppc64] + os: [linux] + libc: [glibc] + + '@rolldown/binding-linux-s390x-gnu@1.2.12': + resolution: {integrity: sha512-qvK4DuAsQc2BSjlx+Xr+IzOIvvxbGZqxFwdWfG6F518Erj0GGISyQbJ6pIappnOxlNPzNHvo/L0BwB30GZ+zVw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [s390x] + os: [linux] + libc: [glibc] + + '@rolldown/binding-linux-x64-gnu@1.2.12': + resolution: {integrity: sha512-Q9uLBO53Xd4QIq1WOycVQyPP1O4HhraEV2qqb3uTrnVw6QZih9duY4vNXOivL1xoUS1/z+W8eF4NMfl2a8Sdjw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [linux] + libc: [glibc] + + '@rolldown/binding-linux-x64-musl@1.2.12': + resolution: {integrity: sha512-3IBxWFMjbOZskDPKv8Lf9BCnahlKuHthWkYnyIxOH/QcJrFcS4EmcenthApkwr/5+nEqZlLzeYbxeMaX7A5u4g==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [linux] + libc: [musl] + + '@rolldown/binding-openharmony-arm64@1.2.12': + resolution: {integrity: sha512-xtX61xg4LKPkPWilZU1ynKClz5Gj4bf74LML4r3eVLWumKnGjoEr1OSHQhMdbBDoYTi+yjrujvpZe2pUnqCrrA==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [openharmony] + + '@rolldown/binding-win32-arm64-msvc@1.2.12': + resolution: {integrity: sha512-At7fPB6PCaIjzgIhEZFxuT+BBFqiQibJDT4d3PhiR3f4E7bbMZF4aKblbFfEM3sETRDd1YiQx/+U/g/B/ou5Ew==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [win32] + + '@rolldown/binding-win32-x64-msvc@1.2.12': + resolution: {integrity: sha512-WIw2haVKwjuYdXkHaoC0mF8Le71TuCBxjrdKqLbJGctbBABj+ClfmNvtbOnzpq3RokNo5+V1qhtSzJyXorsklQ==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [win32] + + '@rolldown/pluginutils@1.0.1': + resolution: {integrity: sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw==} + + '@types/chai@5.2.3': + resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} + + '@types/deep-eql@4.0.2': + resolution: {integrity: sha512-c9h9dVVMigMPc4bwTvC5dxqtqJZwQPePsWjPlpSOnojbor6pGqdk541lfA7AqFQr5pB1BRdq0juY9db81BwyFw==} + + '@types/estree@1.0.9': + resolution: {integrity: sha512-GhdPgy1el4/ImP05X05Uw4cw2/M93BCUmnEvWZNStlCzEKME4Fkk+YpoA5OiHNQmoS7Cafb8Xa3Pya8m1Qrzeg==} + + '@vitest/mocker@5.0.3': + resolution: {integrity: sha512-T8sWAIbkSyAjkwTcaEc3Iu0o9A27X1/kdXrizhZkGuSKScRQtRzclfAMpOTcGdXCsqxeWlpGy3XjqaW8CpLORg==} + peerDependencies: + msw: ^2.4.9 + vite: ^6.0.0 || ^7.0.0 || ^8.0.0 + peerDependenciesMeta: + msw: + optional: true + vite: + optional: true + + '@vitest/spy@5.0.3': + resolution: {integrity: sha512-XhFysQTB8AZ+P4gMi+Lpo99vg2AZi0qKpaB9yXQl37+CaMEAPO3iH/wGVnSyL5MPERiLezpqTVtrR6UZH5GCXg==} + + ajv@8.20.0: + resolution: {integrity: sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==} + + assertion-error@2.0.1: + resolution: {integrity: sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA==} + engines: {node: '>=12'} + + chai@6.3.0: + resolution: {integrity: sha512-XWAtwJ6OHO+tj0EKCs0Y2UamnyOxseZWltU4x2U2wh8g4AigdjwvtUjvLP2tqkA/avxHEtzxNaqGq/YGNwckKg==} + engines: {node: '>=18'} + + detect-libc@2.1.2: + resolution: {integrity: sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==} + engines: {node: '>=8'} + + es-module-lexer@2.3.2: + resolution: {integrity: sha512-poHGpORABojJJucnV9KbOavETW8lBVnphkW77ER5/BQ5Fz7oXSoCNek7IH3vR5nRjdsEz926ibFYX8KtLQmdyw==} + + estree-walker@3.0.3: + resolution: {integrity: sha512-7RUKfXgSMMkzt6ZuXmqapOurLGPPfgj6l9uRZ7lRGolvk0y2yocc35LdcxKC5PQZdn2DMqioAQ2NoWcrTKmm6g==} + + expect-type@1.4.0: + resolution: {integrity: sha512-KfYbmpRm0VbLjEvVa9yGwCi9GI34xvi7A/HXYWQO65CSD2u3MczUJSuwXKFIxlGsgBQizV9q5J9NHj4VG0n+pA==} + engines: {node: '>=12.0.0'} + + fast-check@4.10.2: + resolution: {integrity: sha512-iK2f+YrcmoeGqk6fA0ea2bptcu/itMIm4NfEozq6N25+aG6h7s5HZbB/k1aV7b5w5sFLMCbbtRUsTVR+BgC3xw==} + engines: {node: '>=12.17.0'} + + fast-deep-equal@3.1.3: + resolution: {integrity: sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==} + + fast-uri@3.1.8: + resolution: {integrity: sha512-GZMtZUTNRpOVIECoXwLNZS5xUGE+mVNbTB8h/7Rwh2TFWcBQiPzTgyZi05BF9UMZKkLJv8XBRJTlU7zg8+ZfMg==} + + fdir@6.5.0: + resolution: {integrity: sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==} + engines: {node: '>=12.0.0'} + peerDependencies: + picomatch: ^3 || ^4 + peerDependenciesMeta: + picomatch: + optional: true + + fsevents@2.3.3: + resolution: {integrity: sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==} + engines: {node: ^8.16.0 || ^10.6.0 || >=11.0.0} + os: [darwin] + + json-schema-traverse@1.0.0: + resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==} + + lightningcss-android-arm64@1.33.0: + resolution: {integrity: sha512-gEpRTalKdosp4Bb8qWtc2iOgE5SeIHlpS1up9bFq2wAyYhl1UdTObYiHe98zEM9SQvSoqQZ1IQD0JNpg3Ml5pg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [android] + + lightningcss-darwin-arm64@1.33.0: + resolution: {integrity: sha512-Sciaz8eenNTKn9b3t7+xr0ipTp9YxKQY4npwQ3mrRuL0BAVHBLyZxofhaKBAVtzmtRZ/zTyo0/to4B1uWG/Djg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [darwin] + + lightningcss-darwin-x64@1.33.0: + resolution: {integrity: sha512-Z5UPAxzrjlWNNyGy6i65cJzzvgJ5D3T6wMvs+gWpY9d7qRhANrxqAp6LhxIgZhWEw18RfJTGcRxjuLIBr+m8XQ==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [darwin] + + lightningcss-freebsd-x64@1.33.0: + resolution: {integrity: sha512-QQM/Ti/hQajJwCY+RiWuCZ9sdtI/XQk7nDK5vC8kkdwixezOlDgvDx7+RT+QjK6FcFT4MpsuoBnHIo/O3StRRg==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [freebsd] + + lightningcss-linux-arm-gnueabihf@1.33.0: + resolution: {integrity: sha512-N7FVBe6iS24MlM6R/4RBTxGhQheZGs7tiQ9U32UtF75NzP5Q7xWPRqLBCKxlRQRk3rY1jCIPLzx7WzOhuUIRLQ==} + engines: {node: '>= 12.0.0'} + cpu: [arm] + os: [linux] + + lightningcss-linux-arm64-gnu@1.33.0: + resolution: {integrity: sha512-j2v/itmy4HlNxlc6voKXYgBqNi0Ng2LShg4z7GufpEgs05P+2suBVyi9I6YHq5uoVFx9ETin3eCEhLVyXGQnKg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [linux] + libc: [glibc] + + lightningcss-linux-arm64-musl@1.33.0: + resolution: {integrity: sha512-yiO5ROMuYQgXbC60yjZU5CYSFZGKXL0HFATXt9mHJn1+zW55oCtMI9NfcVhYLMFDL7gV7oBPon/EmMMGg2OvtQ==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [linux] + libc: [musl] + + lightningcss-linux-x64-gnu@1.33.0: + resolution: {integrity: sha512-ar+Ju7LmcN0Jo4FpL4hpFybwNG9/3A/Br5KW2n2jyODg3MEZXaDYADdemoNS+BDNfMgKvylJLj4S5tyRActuAg==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [linux] + libc: [glibc] + + lightningcss-linux-x64-musl@1.33.0: + resolution: {integrity: sha512-RYiYbkokw0trfKqqzfF55lginwEPrD3OJDfTuJzFs1MK6iFnDenaz1fqLLtX4ITG3OktJQXOeTaw1awrBAlZPw==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [linux] + libc: [musl] + + lightningcss-win32-arm64-msvc@1.33.0: + resolution: {integrity: sha512-1K+MPfLSFVpphzpdbfkhlWk6wBrTObBzS2T6db10PNOZgR9GoVsAWzwNyuhUYYbTp23j+4RrncfujZ4uAzXvwA==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [win32] + + lightningcss-win32-x64-msvc@1.33.0: + resolution: {integrity: sha512-OlEICDx/Xl0FqSp4bry8zFnCvGpig3Gl4gCquvYwHuqJKEC1+n9NgDniFvqHGmMv1ZkqDJrDqKKSykTDX+ehuA==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [win32] + + lightningcss@1.33.0: + resolution: {integrity: sha512-WkUDrojuJs0xkgGf2udWxa3yGBRxPtxUkB79i6aCZLRgc7PM8fZe9TosfPDcvEpQZbuFASnHYmRLBLUbmLOIIA==} + engines: {node: '>= 12.0.0'} + + magic-string@1.4.2: + resolution: {integrity: sha512-vG+rjFRj1PqdIBozIxAGMjPlOhaVe+GXpbttY/iSK7rGcJRMlwNJO7dcUwmUqkymsFLJiNGI06t4D7Fr7yRC9g==} + + nanoid@3.3.19: + resolution: {integrity: sha512-Y2tUNy4ouw6tq5oDSKeQYGOyhkUBhNOcGV/02KC+6kd9eDGqdZd++mjMiIDilrBYvjEnCYvVtsuHCuP+okSfug==} + engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} + hasBin: true + + obug@2.2.1: + resolution: {integrity: sha512-XrsrhT5sybtKI6wakr2SPOlGZWWYbUXZ7a0jT8/QOeAPau+1X/bSegNe5YR75oJmEZQbKningirmGOEJCIk61Q==} + engines: {node: '>=12.20.0'} + + picocolors@1.1.1: + resolution: {integrity: sha512-xceH2snhtb5M9liqDsmEw56le376mTZkEX/jEb/RxNFyegNul7eNslCXP9FDj/Lcu0X8KEyMceP2ntpaHrDEVA==} + + picomatch@4.0.7: + resolution: {integrity: sha512-qcJu88Q2IWqJsDD529JKMdwGm/dvInW4HvQnRwiH9JtihJvzGOscDtHE3x1pBKeUOTysQ8kVmLnJ2kJu7yhcGA==} + engines: {node: '>=12'} + + postcss@8.5.28: + resolution: {integrity: sha512-RRuzqDtt5Y9h3quz5hWhK+TPnsmVs6WwSU6LkJMeY4HstUEDuYTG8UJSdawMRzmzAtV+KEoG8N3Qg2qLy5vM/A==} + engines: {node: ^10 || ^12 || >=14} + + pure-rand@8.4.2: + resolution: {integrity: sha512-vvuOGgcuPJAirlHvuQw1TrOiw7ptaIXXmIbNuiNOY6lNGJJH49PQ1Kj4nd783nPdQhQdicgOjVI2yI/9BD6/Ng==} + + require-from-string@2.0.2: + resolution: {integrity: sha512-Xf0nWe6RseziFMu+Ap9biiUbmplq6S9/p+7w7YXP/JBHhrUDDUhwa+vANyubuqfZWTveU//DYVGsDG7RKL/vEw==} + engines: {node: '>=0.10.0'} + + rolldown@1.2.12: + resolution: {integrity: sha512-8wafseiaG80xmXSfqidUNqZcylTlhmPZZt+za2m+js2sFZ8dTNlhIOV2WcbIPx2hgwPBJpEUGFAMZ9bgBBLTSQ==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + + source-map-js@1.2.2: + resolution: {integrity: sha512-KGj/8Y43x35aZVDtt+J4mK1hoLGHULMYfSkODJNQjNDC3oW1PqPoxMwo0pLUsWM/UEGzON/NxeHywEfNXNP3Vw==} + engines: {node: '>=0.10.0'} + + std-env@4.3.0: + resolution: {integrity: sha512-OtU/EgQ1kIm5KwqQpBC6ZEMXrZRui11w8zgfTWp8cdO9B8OaPsbA8bTHO2P+HNo1VlUTGMVBwPhydu6poeXiag==} + + tinybench@6.2.0: + resolution: {integrity: sha512-78U2TlB2CnVenajOFzf3BKSm0J6oz5L0NV7g32LCPccvYc0lbWvys4d3uUUCS2B1N8PAf2+aekR8i1KbC3HO7Q==} + engines: {node: '>=20.0.0'} + + tinyexec@1.3.1: + resolution: {integrity: sha512-GCvB3aoys96IuDFBMcTB46JOR6mdMtAToqwiW8JlWhsoh1mhHi/xn9ss/Dg7N555GiJyEt2qzoG/NHCwM6h1EA==} + engines: {node: '>=18'} + + tinyglobby@0.2.17: + resolution: {integrity: sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g==} + engines: {node: '>=12.0.0'} + + vite@8.3.2: + resolution: {integrity: sha512-SQr1x6W5vVSbROg7vsyXIaxK9b0G7zsT68acdWWRmnBUsgDieLCRG+Rep9WdZgcposvv/GSnr4GUUBqB3vXq6w==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + peerDependencies: + '@types/node': ^20.19.0 || >=22.12.0 + '@vitejs/devtools': ^0.7.1 + esbuild: ^0.27.0 || ^0.28.0 + jiti: '>=1.21.0' + less: ^4.0.0 + sass: ^1.70.0 + sass-embedded: ^1.70.0 + stylus: '>=0.54.8' + sugarss: ^5.0.0 + terser: ^5.16.0 + tsx: ^4.8.1 + yaml: ^2.4.2 + peerDependenciesMeta: + '@types/node': + optional: true + '@vitejs/devtools': + optional: true + esbuild: + optional: true + jiti: + optional: true + less: + optional: true + sass: + optional: true + sass-embedded: + optional: true + stylus: + optional: true + sugarss: + optional: true + terser: + optional: true + tsx: + optional: true + yaml: + optional: true + + vitest@5.0.3: + resolution: {integrity: sha512-xMw97S3rjdtj5dkVat7jCsqWBpvchs3RlpQctUqwJD0KkERk40vz2fJ77lDwW/Vzh/pk18eItYAzkodhSes3jQ==} + engines: {node: ^22.12.0 || ^24.0.0 || >=26.0.0} + hasBin: true + peerDependencies: + '@edge-runtime/vm': '*' + '@opentelemetry/api': ^1.9.0 + '@types/node': ^22.0.0 || >=24.0.0 + '@vitest/browser-playwright': 5.0.3 + '@vitest/browser-preview': 5.0.3 + '@vitest/browser-webdriverio': ^5.0.0-beta.5 || >=5.0.0 + '@vitest/coverage-istanbul': 5.0.3 + '@vitest/coverage-v8': 5.0.3 + '@vitest/ui': 5.0.3 + happy-dom: '*' + jsdom: '*' + vite: ^6.4.0 || ^7.0.0 || ^8.0.0 + peerDependenciesMeta: + '@edge-runtime/vm': + optional: true + '@opentelemetry/api': + optional: true + '@types/node': + optional: true + '@vitest/browser-playwright': + optional: true + '@vitest/browser-preview': + optional: true + '@vitest/browser-webdriverio': + optional: true + '@vitest/coverage-istanbul': + optional: true + '@vitest/coverage-v8': + optional: true + '@vitest/ui': + optional: true + happy-dom: + optional: true + jsdom: + optional: true + + why-is-node-running@3.2.1: + resolution: {integrity: sha512-Tb2FUhB4vUsGQlfSquQLYkApkuPAFQXGFzxWKHHumVz2dK+X1RUm/HnID4+TfIGYJ1kTcwOaCk/buYCEJr6YjQ==} + engines: {node: '>=20.11'} + hasBin: true + +snapshots: + + '@fast-check/vitest@0.5.0(vitest@5.0.3(vite@8.3.2))': + dependencies: + fast-check: 4.10.2 + vitest: 5.0.3(vite@8.3.2) + + '@jridgewell/resolve-uri@3.1.2': {} + + '@jridgewell/sourcemap-codec@1.6.0': {} + + '@jridgewell/trace-mapping@0.3.31': + dependencies: + '@jridgewell/resolve-uri': 3.1.2 + '@jridgewell/sourcemap-codec': 1.6.0 + + '@oxc-project/types@0.152.0': {} + + '@rolldown/binding-android-arm-eabi@1.2.12': + optional: true + + '@rolldown/binding-android-arm64@1.2.12': + optional: true + + '@rolldown/binding-darwin-arm64@1.2.12': + optional: true + + '@rolldown/binding-darwin-x64@1.2.12': + optional: true + + '@rolldown/binding-freebsd-x64@1.2.12': + optional: true + + '@rolldown/binding-linux-arm-gnueabihf@1.2.12': + optional: true + + '@rolldown/binding-linux-arm64-gnu@1.2.12': + optional: true + + '@rolldown/binding-linux-arm64-musl@1.2.12': + optional: true + + '@rolldown/binding-linux-ppc64-gnu@1.2.12': + optional: true + + '@rolldown/binding-linux-s390x-gnu@1.2.12': + optional: true + + '@rolldown/binding-linux-x64-gnu@1.2.12': + optional: true + + '@rolldown/binding-linux-x64-musl@1.2.12': + optional: true + + '@rolldown/binding-openharmony-arm64@1.2.12': + optional: true + + '@rolldown/binding-win32-arm64-msvc@1.2.12': + optional: true + + '@rolldown/binding-win32-x64-msvc@1.2.12': + optional: true + + '@rolldown/pluginutils@1.0.1': {} + + '@types/chai@5.2.3': + dependencies: + '@types/deep-eql': 4.0.2 + assertion-error: 2.0.1 + + '@types/deep-eql@4.0.2': {} + + '@types/estree@1.0.9': {} + + '@vitest/mocker@5.0.3(vite@8.3.2)': + dependencies: + '@jridgewell/trace-mapping': 0.3.31 + '@vitest/spy': 5.0.3 + estree-walker: 3.0.3 + magic-string: 1.4.2 + optionalDependencies: + vite: 8.3.2 + + '@vitest/spy@5.0.3': {} + + ajv@8.20.0: + dependencies: + fast-deep-equal: 3.1.3 + fast-uri: 3.1.8 + json-schema-traverse: 1.0.0 + require-from-string: 2.0.2 + + assertion-error@2.0.1: {} + + chai@6.3.0: {} + + detect-libc@2.1.2: {} + + es-module-lexer@2.3.2: {} + + estree-walker@3.0.3: + dependencies: + '@types/estree': 1.0.9 + + expect-type@1.4.0: {} + + fast-check@4.10.2: + dependencies: + pure-rand: 8.4.2 + + fast-deep-equal@3.1.3: {} + + fast-uri@3.1.8: {} + + fdir@6.5.0(picomatch@4.0.7): + optionalDependencies: + picomatch: 4.0.7 + + fsevents@2.3.3: + optional: true + + json-schema-traverse@1.0.0: {} + + lightningcss-android-arm64@1.33.0: + optional: true + + lightningcss-darwin-arm64@1.33.0: + optional: true + + lightningcss-darwin-x64@1.33.0: + optional: true + + lightningcss-freebsd-x64@1.33.0: + optional: true + + lightningcss-linux-arm-gnueabihf@1.33.0: + optional: true + + lightningcss-linux-arm64-gnu@1.33.0: + optional: true + + lightningcss-linux-arm64-musl@1.33.0: + optional: true + + lightningcss-linux-x64-gnu@1.33.0: + optional: true + + lightningcss-linux-x64-musl@1.33.0: + optional: true + + lightningcss-win32-arm64-msvc@1.33.0: + optional: true + + lightningcss-win32-x64-msvc@1.33.0: + optional: true + + lightningcss@1.33.0: + dependencies: + detect-libc: 2.1.2 + optionalDependencies: + lightningcss-android-arm64: 1.33.0 + lightningcss-darwin-arm64: 1.33.0 + lightningcss-darwin-x64: 1.33.0 + lightningcss-freebsd-x64: 1.33.0 + lightningcss-linux-arm-gnueabihf: 1.33.0 + lightningcss-linux-arm64-gnu: 1.33.0 + lightningcss-linux-arm64-musl: 1.33.0 + lightningcss-linux-x64-gnu: 1.33.0 + lightningcss-linux-x64-musl: 1.33.0 + lightningcss-win32-arm64-msvc: 1.33.0 + lightningcss-win32-x64-msvc: 1.33.0 + + magic-string@1.4.2: + dependencies: + '@jridgewell/sourcemap-codec': 1.6.0 + + nanoid@3.3.19: {} + + obug@2.2.1: {} + + picocolors@1.1.1: {} + + picomatch@4.0.7: {} + + postcss@8.5.28: + dependencies: + nanoid: 3.3.19 + picocolors: 1.1.1 + source-map-js: 1.2.2 + + pure-rand@8.4.2: {} + + require-from-string@2.0.2: {} + + rolldown@1.2.12: + dependencies: + '@oxc-project/types': 0.152.0 + '@rolldown/pluginutils': 1.0.1 + optionalDependencies: + '@rolldown/binding-android-arm-eabi': 1.2.12 + '@rolldown/binding-android-arm64': 1.2.12 + '@rolldown/binding-darwin-arm64': 1.2.12 + '@rolldown/binding-darwin-x64': 1.2.12 + '@rolldown/binding-freebsd-x64': 1.2.12 + '@rolldown/binding-linux-arm-gnueabihf': 1.2.12 + '@rolldown/binding-linux-arm64-gnu': 1.2.12 + '@rolldown/binding-linux-arm64-musl': 1.2.12 + '@rolldown/binding-linux-ppc64-gnu': 1.2.12 + '@rolldown/binding-linux-s390x-gnu': 1.2.12 + '@rolldown/binding-linux-x64-gnu': 1.2.12 + '@rolldown/binding-linux-x64-musl': 1.2.12 + '@rolldown/binding-openharmony-arm64': 1.2.12 + '@rolldown/binding-win32-arm64-msvc': 1.2.12 + '@rolldown/binding-win32-x64-msvc': 1.2.12 + + source-map-js@1.2.2: {} + + std-env@4.3.0: {} + + tinybench@6.2.0: {} + + tinyexec@1.3.1: {} + + tinyglobby@0.2.17: + dependencies: + fdir: 6.5.0(picomatch@4.0.7) + picomatch: 4.0.7 + + vite@8.3.2: + dependencies: + lightningcss: 1.33.0 + picomatch: 4.0.7 + postcss: 8.5.28 + rolldown: 1.2.12 + tinyglobby: 0.2.17 + optionalDependencies: + fsevents: 2.3.3 + + vitest@5.0.3(vite@8.3.2): + dependencies: + '@types/chai': 5.2.3 + '@vitest/mocker': 5.0.3(vite@8.3.2) + chai: 6.3.0 + es-module-lexer: 2.3.2 + expect-type: 1.4.0 + magic-string: 1.4.2 + obug: 2.2.1 + picomatch: 4.0.7 + std-env: 4.3.0 + tinybench: 6.2.0 + tinyexec: 1.3.1 + tinyglobby: 0.2.17 + vite: 8.3.2 + why-is-node-running: 3.2.1 + transitivePeerDependencies: + - msw + + why-is-node-running@3.2.1: {} diff --git a/evals/ratstack-scorecard/pnpm-workspace.yaml b/evals/ratstack-scorecard/pnpm-workspace.yaml new file mode 100644 index 0000000..d728e2b --- /dev/null +++ b/evals/ratstack-scorecard/pnpm-workspace.yaml @@ -0,0 +1,3 @@ +packages: + - . +storeDir: node_modules/.pnpm-store diff --git a/evals/ratstack-scorecard/scorecard.schema.json b/evals/ratstack-scorecard/scorecard.schema.json new file mode 100644 index 0000000..117c072 --- /dev/null +++ b/evals/ratstack-scorecard/scorecard.schema.json @@ -0,0 +1,303 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$id": "https://github.com/systemfsoftware/effect-endgame-starter-kit/evals/ratstack-scorecard/scorecard.schema.json", + "title": "rat-stack scorecard", + "type": "object", + "additionalProperties": false, + "required": ["schemaVersion", "provenance", "rows", "ratchet"], + "properties": { + "schemaVersion": { "const": 1 }, + "provenance": { "$ref": "#/definitions/documentProvenance" }, + "rows": { "type": "array", "items": { "$ref": "#/definitions/row" } }, + "ratchet": { "$ref": "#/definitions/ratchet" } + }, + "definitions": { + "sha": { "type": "string", "minLength": 1 }, + "documentProvenance": { + "type": "object", + "additionalProperties": false, + "required": ["commit", "ratstackCommit", "instrumentHash", "nixpkgsRev", "runner", "generatedAt"], + "properties": { + "commit": { "$ref": "#/definitions/sha" }, + "ratstackCommit": { "$ref": "#/definitions/sha" }, + "instrumentHash": { "$ref": "#/definitions/sha" }, + "nixpkgsRev": { "$ref": "#/definitions/sha" }, + "runner": { "type": "string", "minLength": 1 }, + "generatedAt": { "type": "string", "minLength": 1 } + } + }, + "cellProvenance": { + "type": "object", + "additionalProperties": false, + "required": ["side", "commit", "instrumentHash", "nixpkgsRev", "runner", "measuredAt", "tools"], + "properties": { + "side": { "enum": ["ratstack", "starter"] }, + "commit": { "$ref": "#/definitions/sha" }, + "instrumentHash": { "$ref": "#/definitions/sha" }, + "nixpkgsRev": { "$ref": "#/definitions/sha" }, + "runner": { "type": "string", "minLength": 1 }, + "measuredAt": { "type": "string", "minLength": 1 }, + "tools": { "type": "object", "additionalProperties": { "type": "string" } }, + "liveCommit": { "$ref": "#/definitions/sha" } + } + }, + "citation": { + "type": "object", + "additionalProperties": false, + "required": ["file", "lines", "text"], + "properties": { + "file": { "type": "string", "minLength": 1 }, + "lines": { + "type": "array", + "items": [{ "type": "integer", "minimum": 1 }, { "type": "integer", "minimum": 1 }], + "minItems": 2, + "maxItems": 2 + }, + "text": { "type": "string", "minLength": 1 } + } + }, + "cell": { + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "runs"], + "properties": { + "_tag": { "const": "Measured" }, + "runs": { "type": "array", "items": { "type": "number" } } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "reason"], + "properties": { "_tag": { "const": "Absent" }, "reason": { "type": "string" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "sha"], + "properties": { "_tag": { "const": "NoDeployment" }, "sha": { "$ref": "#/definitions/sha" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "citation", "check"], + "properties": { + "_tag": { "const": "Unsupported" }, + "citation": { "$ref": "#/definitions/citation" }, + "check": { + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["_tag"], + "properties": { "_tag": { "const": "Verified" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "found"], + "properties": { "_tag": { "const": "Contradicted" }, "found": { "type": "string" } } + } + ] + } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "error"], + "properties": { "_tag": { "const": "Unmeasurable" }, "error": { "type": "string" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "name"], + "properties": { "_tag": { "const": "NoSecret" }, "name": { "type": "string" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "error"], + "properties": { "_tag": { "const": "InstrumentError" }, "error": { "type": "string" } } + } + ] + }, + "measuredCell": { + "type": "object", + "additionalProperties": false, + "required": ["cell", "provenance"], + "properties": { + "cell": { "$ref": "#/definitions/cell" }, + "provenance": { "$ref": "#/definitions/cellProvenance" } + } + }, + "verdict": { + "oneOf": [ + { "$ref": "#/definitions/bareTag/Beaten" }, + { "$ref": "#/definitions/bareTag/NotBeaten" }, + { "$ref": "#/definitions/bareTag/Tie" }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "error"], + "properties": { "_tag": { "const": "InstrumentError" }, "error": { "type": "string" } } + } + ] + }, + "bareTag": { + "Beaten": { + "type": "object", + "additionalProperties": false, + "required": ["_tag"], + "properties": { "_tag": { "const": "Beaten" } } + }, + "NotBeaten": { + "type": "object", + "additionalProperties": false, + "required": ["_tag"], + "properties": { "_tag": { "const": "NotBeaten" } } + }, + "Tie": { + "type": "object", + "additionalProperties": false, + "required": ["_tag"], + "properties": { "_tag": { "const": "Tie" } } + } + }, + "flag": { + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "live", "pin"], + "properties": { + "_tag": { "const": "LiveDiffersFromPin" }, + "live": { "$ref": "#/definitions/sha" }, + "pin": { "$ref": "#/definitions/sha" } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "primary", "crossCheck"], + "properties": { + "_tag": { "const": "ScannersDisagree" }, + "primary": { "type": "number" }, + "crossCheck": { "type": "number" } + } + } + ] + }, + "rowDefinition": { + "type": "object", + "additionalProperties": false, + "required": ["id", "metric", "bin", "label", "unit", "direction", "kind", "runs", "family"], + "properties": { + "id": { "type": "string", "pattern": "^M[0-9]+(\\.[a-z0-9-]+)?$" }, + "metric": { "type": "string", "pattern": "^M[0-9]+$" }, + "bin": { "type": "string", "minLength": 1 }, + "label": { "type": "string", "minLength": 1 }, + "unit": { "type": "string", "minLength": 1 }, + "direction": { "enum": ["lower", "higher"] }, + "kind": { "enum": ["count", "measurement"] }, + "runs": { "type": "integer", "minimum": 1 }, + "family": { "enum": ["static", "cold-path", "gate-mutation", "running-stack", "agent-surfaces", "networked"] } + } + }, + "row": { + "type": "object", + "additionalProperties": false, + "required": ["definition", "definitionHash", "ratstack", "starter", "verdict", "flags"], + "properties": { + "definition": { "$ref": "#/definitions/rowDefinition" }, + "definitionHash": { "$ref": "#/definitions/sha" }, + "ratstack": { "$ref": "#/definitions/measuredCell" }, + "starter": { "$ref": "#/definitions/measuredCell" }, + "verdict": { "$ref": "#/definitions/verdict" }, + "flags": { "type": "array", "items": { "$ref": "#/definitions/flag" } } + } + }, + "cellTag": { + "enum": ["Measured", "Absent", "NoDeployment", "Unsupported", "Unmeasurable", "NoSecret", "InstrumentError"] + }, + "outcome": { + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "id"], + "properties": { "_tag": { "enum": ["Held", "New", "ReBaselined"] }, "id": { "type": "string" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "id", "cause"], + "properties": { + "_tag": { "const": "Neutral" }, + "id": { "type": "string" }, + "cause": { "$ref": "#/definitions/cellTag" } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "id", "error"], + "properties": { + "_tag": { "const": "InstrumentError" }, + "id": { "type": "string" }, + "error": { "type": "string" } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "id", "ratstack", "starter"], + "properties": { + "_tag": { "const": "LostBeaten" }, + "id": { "type": "string" }, + "ratstack": { "$ref": "#/definitions/cellTag" }, + "starter": { "$ref": "#/definitions/cellTag" } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "id", "main", "pr"], + "properties": { + "_tag": { "const": "Regressed" }, + "id": { "type": "string" }, + "main": { "type": "string" }, + "pr": { "type": "string" } + } + } + ] + }, + "ratchet": { + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "outcomes", "failures"], + "properties": { + "_tag": { "const": "FirstBaseline" }, + "outcomes": { "type": "array", "items": { "$ref": "#/definitions/outcome" } }, + "failures": { "type": "array", "items": { "$ref": "#/definitions/outcome" } } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "mainCommit", "outcomes", "failures"], + "properties": { + "_tag": { "const": "Compared" }, + "mainCommit": { "$ref": "#/definitions/sha" }, + "outcomes": { "type": "array", "items": { "$ref": "#/definitions/outcome" } }, + "failures": { "type": "array", "items": { "$ref": "#/definitions/outcome" } } + } + } + ] + } + } +} diff --git a/evals/ratstack-scorecard/src/metrics/registry.ts b/evals/ratstack-scorecard/src/metrics/registry.ts new file mode 100644 index 0000000..c3dfbea --- /dev/null +++ b/evals/ratstack-scorecard/src/metrics/registry.ts @@ -0,0 +1,377 @@ +import type { Family, RowDefinition } from '../model/cell.ts' + +const count3 = { kind: 'count', runs: 3 } as const +const measured5 = { kind: 'measurement', runs: 5 } as const +const measured3 = { kind: 'measurement', runs: 3 } as const + +export const familyTimeoutMinutes: Readonly> = { + 'static': 20, + 'cold-path': 120, + 'gate-mutation': 240, + 'running-stack': 90, + 'agent-surfaces': 90, + 'networked': 60, +} + +export const rowDefinitions: readonly RowDefinition[] = [ + { + id: 'M1.checks', + metric: 'M1', + bin: 'apps/mischief front door', + label: 'URL Scanner agentReadiness checks passed', + unit: 'checks', + direction: 'higher', + ...count3, + family: 'networked', + }, + { + id: 'M1.level', + metric: 'M1', + bin: 'apps/mischief front door', + label: 'URL Scanner agentReadiness level (0-5)', + unit: 'level', + direction: 'higher', + ...count3, + family: 'networked', + }, + { + id: 'M2', + metric: 'M2', + bin: 'apps/mischief front door', + label: 'isitagentready /api/scan level (cross-check)', + unit: 'level', + direction: 'higher', + ...count3, + family: 'networked', + }, + { + id: 'M3.performance', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Lighthouse performance score on /', + unit: 'score', + direction: 'higher', + ...measured5, + family: 'networked', + }, + { + id: 'M3.lcp', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Largest Contentful Paint on /', + unit: 'ms', + direction: 'lower', + ...measured5, + family: 'networked', + }, + { + id: 'M3.cls', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Cumulative Layout Shift on /', + unit: 'CLS', + direction: 'lower', + ...measured5, + family: 'networked', + }, + { + id: 'M3.tbt', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Total Blocking Time on /', + unit: 'ms', + direction: 'lower', + ...measured5, + family: 'networked', + }, + { + id: 'M3.bytes', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Total byte weight of /', + unit: 'bytes', + direction: 'lower', + ...measured5, + family: 'networked', + }, + { + id: 'M3.a11y', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Failed accessibility audits on /', + unit: 'audits', + direction: 'lower', + ...measured5, + family: 'networked', + }, + { + id: 'M4', + metric: 'M4', + bin: '/pins.md', + label: 'Versions on the served pins page that differ from the lockfile', + unit: 'versions', + direction: 'lower', + ...count3, + family: 'agent-surfaces', + }, + { + id: 'M5', + metric: 'M5', + bin: '/log.md', + label: 'Share of change-log entries linking both a PR and a CI run', + unit: 'share', + direction: 'higher', + ...count3, + family: 'agent-surfaces', + }, + { + id: 'M6', + metric: 'M6', + bin: 'apps/web', + label: 'Gzipped client JS emitted by the web build', + unit: 'bytes', + direction: 'lower', + ...count3, + family: 'cold-path', + }, + { + id: 'M7', + metric: 'M7', + bin: 'apps/mischief Worker', + label: 'Gzipped Worker script emitted by the deploy build', + unit: 'bytes', + direction: 'lower', + ...count3, + family: 'cold-path', + }, + { + id: 'M8', + metric: 'M8', + bin: 'packages/capability', + label: 'Mean projections per read capability whose output equals HTTP', + unit: 'projections', + direction: 'higher', + ...count3, + family: 'agent-surfaces', + }, + { + id: 'M9', + metric: 'M9', + bin: 'apps/cli', + label: 'CLI process start to a correct answer for one read capability', + unit: 'ms', + direction: 'lower', + ...measured5, + family: 'agent-surfaces', + }, + { + id: 'M10.seconds', + metric: 'M10', + bin: 'MCP + agent front door', + label: 'Claude Code seconds to a correct answer through MCP only (median of 15)', + unit: 's', + direction: 'lower', + ...measured5, + family: 'agent-surfaces', + }, + { + id: 'M10.correct', + metric: 'M10', + bin: 'MCP + agent front door', + label: 'Claude Code correctness rate through MCP only', + unit: 'share', + direction: 'higher', + ...measured5, + family: 'agent-surfaces', + }, + { + id: 'M11', + metric: 'M11', + bin: 'packages/core + packages/intake-live', + label: 'Successes above 1 when 300 concurrent requests redeem one single-use token', + unit: 'oversells', + direction: 'lower', + ...count3, + family: 'running-stack', + }, + { + id: 'M12', + metric: 'M12', + bin: 'packages/core lifecycles', + label: 'Seats granted beyond capacity for 300 concurrent claims on 100 seats', + unit: 'seats', + direction: 'lower', + ...count3, + family: 'running-stack', + }, + { + id: 'M13', + metric: 'M13', + bin: 'packages/database', + label: "Share of declared store adapters passing the side's own store suite", + unit: 'share', + direction: 'higher', + ...count3, + family: 'gate-mutation', + }, + { + id: 'M14', + metric: 'M14', + bin: 'packages/auth', + label: 'Auth endpoints that accept a password credential', + unit: 'endpoints', + direction: 'lower', + ...count3, + family: 'running-stack', + }, + { + id: 'M15', + metric: 'M15', + bin: 'packages/events', + label: 'Distinct cookies set on anonymous GETs of every sitemap page', + unit: 'cookies', + direction: 'lower', + ...count3, + family: 'agent-surfaces', + }, + { + id: 'M16', + metric: 'M16', + bin: 'packages/lore', + label: 'Lore operation × surface pairs returning a schema-valid result', + unit: 'pairs', + direction: 'higher', + ...count3, + family: 'agent-surfaces', + }, + { + id: 'M17', + metric: 'M17', + bin: 'packages/subscriber-delivery', + label: 'Requests without exactly one confirmation after a mid-burst kill and restart', + unit: 'requests', + direction: 'lower', + ...count3, + family: 'running-stack', + }, + { + id: 'M18', + metric: 'M18', + bin: 'packages/devtools', + label: 'Share of 20 scripted calls retrievable from the local devtools store', + unit: 'share', + direction: 'higher', + ...count3, + family: 'agent-surfaces', + }, + { + id: 'M19', + metric: 'M19', + bin: 'packages/intake-live abuse bounds', + label: "Requests accepted beyond the side's declared per-email bound", + unit: 'requests', + direction: 'lower', + ...count3, + family: 'running-stack', + }, + { + id: 'M20', + metric: 'M20', + bin: 'packages/code-snippets', + label: 'Build fails after a symbol quoted on a page is renamed (1 or 0)', + unit: 'caught', + direction: 'higher', + ...count3, + family: 'gate-mutation', + }, + { + id: 'M21', + metric: 'M21', + bin: 'apps/infra', + label: 'Outbound connection attempts while importing the stack module', + unit: 'attempts', + direction: 'lower', + ...count3, + family: 'gate-mutation', + }, + { + id: 'M22', + metric: 'M22', + bin: 'apps/infra', + label: 'Removing one binding from the stack fails typecheck (1 or 0)', + unit: 'caught', + direction: 'higher', + ...count3, + family: 'gate-mutation', + }, + { + id: 'M23', + metric: 'M23', + bin: 'fence: debt ledger', + label: 'Suppression directives in tracked source', + unit: 'directives', + direction: 'lower', + ...count3, + family: 'static', + }, + { + id: 'M24', + metric: 'M24', + bin: 'fence: tests', + label: "Sabotages of shared published behaviours that turn the side's gate red", + unit: 'share', + direction: 'higher', + ...count3, + family: 'gate-mutation', + }, + { + id: 'M25', + metric: 'M25', + bin: 'keep-or-cut / bin removal', + label: 'Share of documented bin removals after which the gate stays green', + unit: 'share', + direction: 'higher', + ...count3, + family: 'gate-mutation', + }, + { + id: 'M26', + metric: 'M26', + bin: 'acceptance-cold-clone.sh', + label: 'Fresh clone to green documented gate', + unit: 's', + direction: 'lower', + ...measured3, + family: 'cold-path', + }, + { + id: 'M27', + metric: 'M27', + bin: 'install', + label: 'HTTP requests during the cold install', + unit: 'requests', + direction: 'lower', + ...count3, + family: 'cold-path', + }, + { + id: 'M28', + metric: 'M28', + bin: 'install', + label: "Distinct name@version packages in the side's lockfile", + unit: 'packages', + direction: 'lower', + ...count3, + family: 'static', + }, + { + id: 'M29', + metric: 'M29', + bin: 'skills/* + /.well-known/agent-skills', + label: 'Share of served skills that `skills add` installs with valid frontmatter', + unit: 'share', + direction: 'higher', + ...count3, + family: 'agent-surfaces', + }, +] diff --git a/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts b/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts new file mode 100644 index 0000000..751f8fc --- /dev/null +++ b/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, test } from 'vitest' +import type { Cell, MeasuredCell, Row, RowDefinition } from './cell.ts' +import { compareWithMain } from './compare-with-main.workflow.ts' +import { judgeRow } from './judge-row.workflow.ts' + +const lcp: RowDefinition = { + id: 'M3.lcp', + metric: 'M3', + bin: 'apps/mischief home page', + label: 'Largest Contentful Paint on /', + unit: 'ms', + direction: 'lower', + kind: 'measurement', + runs: 5, + family: 'networked', +} +const debt: RowDefinition = { ...lcp, id: 'M23', metric: 'M23', kind: 'count', runs: 3, family: 'static' } +const passwords: RowDefinition = { ...debt, id: 'M14', metric: 'M14', family: 'running-stack' } +const cli: RowDefinition = { ...lcp, id: 'M9', metric: 'M9', family: 'agent-surfaces' } + +const measured = (runs: readonly number[]): Cell => ({ _tag: 'Measured', runs }) + +const at = (cell: Cell): MeasuredCell => ({ + cell, + provenance: { + side: 'starter', + commit: 'a'.repeat(40), + instrumentHash: 'b'.repeat(40), + nixpkgsRev: 'c'.repeat(40), + runner: 'fleet-large-1', + measuredAt: '2026-10-05T00:00:00.000Z', + tools: {}, + }, +}) + +const row = (definition: RowDefinition, ratstack: Cell, starter: Cell): Row => ({ + definition, + definitionHash: 'd'.repeat(40), + ratstack: at(ratstack), + starter: at(starter), + verdict: judgeRow({ definition, ratstack, starter }), + flags: [], +}) + +const seats = { + file: 'packages/core/src/join-interest-contract.ts', + lines: [22, 26] as const, + text: 'Confirmation is not a seat', +} + +describe('plan acceptance examples', () => { + test('AE1: overlapping LCP ranges are not beaten; a starter whose worst run is 409 is', () => { + const ratstack = measured([410, 420, 455, 430, 418]) + expect(judgeRow({ definition: lcp, ratstack, starter: measured([380, 395, 412, 390, 401]) })._tag).toBe( + 'NotBeaten', + ) + expect(judgeRow({ definition: lcp, ratstack, starter: measured([380, 395, 409, 390, 401]) })._tag).toBe('Beaten') + }) + + test('AE2: count runs of 217, 217, 216 are an instrument error', () => { + expect(judgeRow({ definition: debt, ratstack: measured([217, 217, 216]), starter: measured([3, 3, 3]) })._tag) + .toBe('InstrumentError') + }) + + test('AE3: the seat citation no longer holding is an instrument error; holding, a measured starter beats it', () => { + const starter = measured([0, 0, 0]) + const contradicted: Cell = { + _tag: 'Unsupported', + citation: seats, + check: { _tag: 'Contradicted', found: 'export const seats = confirmations' }, + } + expect(judgeRow({ definition: debt, ratstack: contradicted, starter })._tag).toBe('InstrumentError') + expect(judgeRow({ definition: debt, ratstack: { ...contradicted, check: { _tag: 'Verified' } }, starter })._tag) + .toBe('Beaten') + }) + + test('AE4: main beats rat-stack on M14 and a PR re-enabling a password route fails, naming M14', () => { + const main = row(passwords, measured([1, 1, 1]), measured([0, 0, 0])) + const pr = row(passwords, measured([1, 1, 1]), measured([1, 1, 1])) + const ratchet = compareWithMain({ rows: [pr], main: { _tag: 'Found', commit: 'e'.repeat(40), rows: [main] } }) + expect(ratchet.failures).toEqual([{ _tag: 'LostBeaten', id: 'M14', ratstack: 'Measured', starter: 'Measured' }]) + }) + + test("AE5: a PR median of 139 ms is inside main's M9 band and passes; 141 ms fails", () => { + const ratstack = measured([90, 91, 92, 93, 94]) + const main = row(cli, ratstack, measured([120, 131, 140, 125, 128])) + const prWith = (runs: readonly number[]) => + compareWithMain({ + rows: [row(cli, ratstack, measured(runs))], + main: { _tag: 'Found', commit: 'e'.repeat(40), rows: [main] }, + }).failures.map((f) => [f._tag, f.id]) + expect(prWith([130, 135, 139, 140, 141])).toEqual([]) + expect(prWith([130, 135, 141, 142, 143])).toEqual([['Regressed', 'M9']]) + }) + + test('AE7: a fork PR with no preview leaves the starter absent and the ratchet neutral', () => { + const scanner: RowDefinition = { ...debt, id: 'M1.level', metric: 'M1', family: 'networked' } + const main = row(scanner, measured([5, 5, 5]), measured([5, 5, 5])) + const pr = row(scanner, measured([5, 5, 5]), { _tag: 'NoDeployment', sha: 'f'.repeat(40) }) + const ratchet = compareWithMain({ rows: [pr], main: { _tag: 'Found', commit: 'e'.repeat(40), rows: [main] } }) + expect(ratchet.failures).toEqual([]) + expect(ratchet.outcomes).toEqual([{ _tag: 'Neutral', id: 'M1.level', cause: 'NoDeployment' }]) + }) +}) diff --git a/evals/ratstack-scorecard/src/model/cell.ts b/evals/ratstack-scorecard/src/model/cell.ts new file mode 100644 index 0000000..aaec3a6 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/cell.ts @@ -0,0 +1,125 @@ +export type Side = 'ratstack' | 'starter' +export type Direction = 'lower' | 'higher' +export type Kind = 'count' | 'measurement' + +export type Family = + | 'static' + | 'cold-path' + | 'gate-mutation' + | 'running-stack' + | 'agent-surfaces' + | 'networked' + +export interface RowDefinition { + readonly id: string + readonly metric: string + readonly bin: string + readonly label: string + readonly unit: string + readonly direction: Direction + readonly kind: Kind + readonly runs: number + readonly family: Family +} + +export interface Citation { + readonly file: string + readonly lines: readonly [number, number] + readonly text: string +} + +export type CitationCheck = + | { readonly _tag: 'Verified' } + | { readonly _tag: 'Contradicted'; readonly found: string } + +export type Cell = + | { readonly _tag: 'Measured'; readonly runs: readonly number[] } + | { readonly _tag: 'Absent'; readonly reason: string } + | { readonly _tag: 'NoDeployment'; readonly sha: string } + | { readonly _tag: 'Unsupported'; readonly citation: Citation; readonly check: CitationCheck } + | { readonly _tag: 'Unmeasurable'; readonly error: string } + | { readonly _tag: 'NoSecret'; readonly name: string } + | { readonly _tag: 'InstrumentError'; readonly error: string } + +export type CellTag = Cell['_tag'] + +export type Verdict = + | { readonly _tag: 'Beaten' } + | { readonly _tag: 'NotBeaten' } + | { readonly _tag: 'Tie' } + | { readonly _tag: 'InstrumentError'; readonly error: string } + +export type VerdictTag = Verdict['_tag'] + +export interface CellProvenance { + readonly side: Side + readonly commit: string + readonly instrumentHash: string + readonly nixpkgsRev: string + readonly runner: string + readonly measuredAt: string + readonly tools: Readonly> + readonly liveCommit?: string +} + +export interface MeasuredCell { + readonly cell: Cell + readonly provenance: CellProvenance +} + +export type Flag = + | { readonly _tag: 'LiveDiffersFromPin'; readonly live: string; readonly pin: string } + | { readonly _tag: 'ScannersDisagree'; readonly primary: number; readonly crossCheck: number } + +export interface Row { + readonly definition: RowDefinition + readonly definitionHash: string + readonly ratstack: MeasuredCell + readonly starter: MeasuredCell + readonly verdict: Verdict + readonly flags: readonly Flag[] +} + +export type RowOutcome = + | { readonly _tag: 'Held'; readonly id: string } + | { readonly _tag: 'New'; readonly id: string } + | { readonly _tag: 'ReBaselined'; readonly id: string } + | { readonly _tag: 'Neutral'; readonly id: string; readonly cause: CellTag } + | { readonly _tag: 'InstrumentError'; readonly id: string; readonly error: string } + | { readonly _tag: 'LostBeaten'; readonly id: string; readonly ratstack: CellTag; readonly starter: CellTag } + | { readonly _tag: 'Regressed'; readonly id: string; readonly main: string; readonly pr: string } + +export type RowOutcomeTag = RowOutcome['_tag'] + +export type Ratchet = + | { + readonly _tag: 'FirstBaseline' + readonly outcomes: readonly RowOutcome[] + readonly failures: readonly RowOutcome[] + } + | { + readonly _tag: 'Compared' + readonly mainCommit: string + readonly outcomes: readonly RowOutcome[] + readonly failures: readonly RowOutcome[] + } + +export type MainBaseline = + | { readonly _tag: 'Missing' } + | { readonly _tag: 'Found'; readonly commit: string; readonly rows: readonly Row[] } + +export interface DocumentProvenance { + readonly commit: string + readonly ratstackCommit: string + readonly instrumentHash: string + readonly nixpkgsRev: string + readonly runner: string + readonly generatedAt: string +} + +export interface ScorecardDocument { + readonly schemaVersion: 1 + readonly provenance: DocumentProvenance + readonly rows: readonly Row[] + readonly ratchet: Ratchet +} diff --git a/evals/ratstack-scorecard/src/model/compare-with-main.workflow.property.test.ts b/evals/ratstack-scorecard/src/model/compare-with-main.workflow.property.test.ts new file mode 100644 index 0000000..e96db22 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/compare-with-main.workflow.property.test.ts @@ -0,0 +1,101 @@ +import { fc, test } from '@fast-check/vitest' +import { describe } from 'vitest' +import type { Cell, Row, Verdict } from './cell.ts' +import { compareWithMain } from './compare-with-main.workflow.ts' +import { row, runsFor, sha } from './scorecard.arbitrary.ts' + +const passing = (verdict: Verdict): boolean => verdict._tag !== 'InstrumentError' + +const withStarter = (r: Row, cell: Cell): Row => ({ ...r, starter: { ...r.starter, cell } }) + +const ids = (rows: readonly { readonly id: string }[]): string => rows.map((r) => r.id).sort().join(',') + +const uniqueRows = fc.uniqueArray(row(), { selector: (r) => r.definition.id, maxLength: 16 }) + +describe('compareWithMain', () => { + test.prop([uniqueRows, sha])('rows compared with themselves fail only where the verdict is an instrument error', ( + rows, + commit, + ) => { + const ratchet = compareWithMain({ rows, main: { _tag: 'Found', commit, rows } }) + return ids(ratchet.failures) === ids(rows.filter((r) => !passing(r.verdict)).map((r) => r.definition)) + }) + + test.prop([uniqueRows])('without a main artifact the ratchet is a first baseline failing only instrument errors', ( + rows, + ) => { + const ratchet = compareWithMain({ rows, main: { _tag: 'Missing' } }) + return ratchet._tag === 'FirstBaseline' && + ids(ratchet.failures) === ids(rows.filter((r) => !passing(r.verdict)).map((r) => r.definition)) + }) + + test.prop([row(), fc.constantFrom({ _tag: 'NotBeaten' }, { _tag: 'Tie' }), sha])( + 'a row beaten on main and not beaten on the PR fails as lost', + (main, prVerdict, commit) => { + const pr = { ...main, verdict: prVerdict } + const ratchet = compareWithMain({ + rows: [pr], + main: { _tag: 'Found', commit, rows: [{ ...main, verdict: { _tag: 'Beaten' } }] }, + }) + return ratchet.failures.length === 1 && ratchet.failures[0]!._tag === 'LostBeaten' + }, + ) + + test.prop([row(), sha, sha])('a changed metric definition is re-baselined and never fails', (main, hash, commit) => { + fc.pre(hash !== main.definitionHash) + const pr = { ...main, definitionHash: hash, verdict: { _tag: 'NotBeaten' } as const } + const ratchet = compareWithMain({ + rows: [pr], + main: { _tag: 'Found', commit, rows: [{ ...main, verdict: { _tag: 'Beaten' } }] }, + }) + return ratchet.failures.length === 0 && ratchet.outcomes[0]!._tag === 'ReBaselined' + }) + + test.prop([ + row().chain((r) => fc.tuple(fc.constant(r), runsFor(r.definition))), + fc.constantFrom( + { _tag: 'Absent', reason: 'bin removed' }, + { _tag: 'Unmeasurable', error: 'gate red at baseline' }, + ), + sha, + ])('a starter value present on main and gone on the PR regresses', ([base, runs], gone, commit) => { + const main = withStarter({ ...base, verdict: { _tag: 'NotBeaten' } }, { _tag: 'Measured', runs }) + const pr = withStarter(main, gone) + const ratchet = compareWithMain({ rows: [pr], main: { _tag: 'Found', commit, rows: [main] } }) + return ratchet.failures.length === 1 && ratchet.failures[0]!._tag === 'Regressed' + }) + + test.prop([ + row().chain((r) => fc.tuple(fc.constant(r), runsFor(r.definition))), + fc.constantFrom({ _tag: 'NoSecret', name: 'ANTHROPIC_API_KEY' }, { _tag: 'NoDeployment', sha: 'abc1234' }), + sha, + ])('a starter value missing for a secret or a fork deployment is neutral', ([base, runs], missing, commit) => { + const main = withStarter({ ...base, verdict: { _tag: 'NotBeaten' } }, { _tag: 'Measured', runs }) + const pr = withStarter(main, missing) + const ratchet = compareWithMain({ rows: [pr], main: { _tag: 'Found', commit, rows: [main] } }) + return ratchet.failures.length === 0 && ratchet.outcomes[0]!._tag === 'Neutral' + }) + + test.prop([ + row({ kind: 'measurement' }).chain((r) => fc.tuple(fc.constant(r), runsFor(r.definition))), + fc.double({ min: 0, max: 1, noNaN: true }), + fc.double({ min: 1e-3, max: 1e3, noNaN: true }), + sha, + ])( + "a PR median inside main's run range holds and one past main's worst run regresses", + ([base, mainRuns], position, beyond, commit) => { + const lo = Math.min(...mainRuns) + const hi = Math.max(...mainRuns) + const worst = { lower: hi, higher: lo }[base.definition.direction] + const worse = { lower: 1, higher: -1 }[base.definition.direction] + const inside = Math.min(hi, Math.max(lo, lo + (hi - lo) * position)) + const main = withStarter({ ...base, verdict: { _tag: 'NotBeaten' } }, { _tag: 'Measured', runs: mainRuns }) + const at = (value: number) => + compareWithMain({ + rows: [withStarter(main, { _tag: 'Measured', runs: mainRuns.map(() => value) })], + main: { _tag: 'Found', commit, rows: [main] }, + }).failures.map((f) => f._tag).join(',') + return at(inside) === '' && at(worst + worse * beyond) === 'Regressed' + }, + ) +}) diff --git a/evals/ratstack-scorecard/src/model/compare-with-main.workflow.ts b/evals/ratstack-scorecard/src/model/compare-with-main.workflow.ts new file mode 100644 index 0000000..f313ae2 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/compare-with-main.workflow.ts @@ -0,0 +1,110 @@ +import type { Cell, Direction, MainBaseline, Ratchet, Row, RowOutcome, Verdict } from './cell.ts' +import { firstRule, matchBaseline, matchCell, matchOutcome, matchVerdict } from './dispatch.ts' +import { median, smallerIsBetter } from './runs.ts' + +export interface CompareWithMainInput { + readonly rows: readonly Row[] + readonly main: MainBaseline +} + +const held = (id: string): RowOutcome => ({ _tag: 'Held', id }) + +const beatenWeight = (verdict: Verdict): number => + matchVerdict(verdict, { Beaten: () => 1, NotBeaten: () => 0, Tie: () => 0, InstrumentError: () => 0 }) + +const describe = (cell: Cell): string => + matchCell(cell, { + Measured: (c) => `runs ${c.runs.join(', ')}`, + Absent: (c) => `absent: ${c.reason}`, + NoDeployment: (c) => `no deployment for ${c.sha}`, + Unsupported: () => 'unsupported', + Unmeasurable: (c) => `unmeasurable: ${c.error}`, + NoSecret: (c) => `no secret ${c.name}`, + InstrumentError: (c) => `instrument error: ${c.error}`, + }) + +const regressed = (main: Row, pr: Row): RowOutcome => ({ + _tag: 'Regressed', + id: pr.definition.id, + main: describe(main.starter.cell), + pr: describe(pr.starter.cell), +}) + +const runsRegressed = (direction: Direction, main: readonly number[], pr: readonly number[]): boolean => + median(smallerIsBetter(direction, pr)) > Math.max(...smallerIsBetter(direction, main)) + +const starterAgainstMainMeasurement = (main: Row, mainRuns: readonly number[], pr: Row): RowOutcome => + matchCell(pr.starter.cell, { + Measured: (cell) => + firstRule( + [[runsRegressed(pr.definition.direction, mainRuns, cell.runs), () => regressed(main, pr)]], + () => held(pr.definition.id), + ), + Absent: () => regressed(main, pr), + Unsupported: () => regressed(main, pr), + Unmeasurable: () => regressed(main, pr), + NoDeployment: (cell) => ({ _tag: 'Neutral', id: pr.definition.id, cause: cell._tag }), + NoSecret: (cell) => ({ _tag: 'Neutral', id: pr.definition.id, cause: cell._tag }), + InstrumentError: (cell) => ({ _tag: 'InstrumentError', id: pr.definition.id, error: cell.error }), + }) + +const starterAgainstMain = (main: Row, pr: Row): RowOutcome => + matchCell(main.starter.cell, { + Measured: (cell) => starterAgainstMainMeasurement(main, cell.runs, pr), + Absent: () => held(pr.definition.id), + NoDeployment: () => held(pr.definition.id), + Unsupported: () => held(pr.definition.id), + Unmeasurable: () => held(pr.definition.id), + NoSecret: () => held(pr.definition.id), + InstrumentError: () => held(pr.definition.id), + }) + +const compareRow = (main: Row, pr: Row): RowOutcome => + firstRule([ + [main.definitionHash !== pr.definitionHash, () => ({ _tag: 'ReBaselined', id: pr.definition.id })], + [ + beatenWeight(main.verdict) > beatenWeight(pr.verdict), + () => ({ + _tag: 'LostBeaten', + id: pr.definition.id, + ratstack: pr.ratstack.cell._tag, + starter: pr.starter.cell._tag, + }), + ], + ], () => starterAgainstMain(main, pr)) + +const unlessInstrumentError = (pr: Row, otherwise: () => RowOutcome): RowOutcome => + matchVerdict(pr.verdict, { + InstrumentError: (verdict) => ({ _tag: 'InstrumentError', id: pr.definition.id, error: verdict.error }), + Beaten: otherwise, + NotBeaten: otherwise, + Tie: otherwise, + }) + +const againstMainRows = (pr: Row, mainRows: readonly Row[]): RowOutcome => + mainRows + .filter((main) => main.definition.id === pr.definition.id) + .reduce((_, main) => compareRow(main, pr), { _tag: 'New', id: pr.definition.id }) + +const fails = (outcome: RowOutcome): boolean => + matchOutcome(outcome, { + Held: () => false, + New: () => false, + ReBaselined: () => false, + Neutral: () => false, + InstrumentError: () => true, + LostBeaten: () => true, + Regressed: () => true, + }) + +export const compareWithMain = (input: CompareWithMainInput): Ratchet => + matchBaseline(input.main, { + Missing: () => { + const outcomes = input.rows.map((pr) => unlessInstrumentError(pr, () => ({ _tag: 'New', id: pr.definition.id }))) + return { _tag: 'FirstBaseline', outcomes, failures: outcomes.filter(fails) } + }, + Found: (main) => { + const outcomes = input.rows.map((pr) => unlessInstrumentError(pr, () => againstMainRows(pr, main.rows))) + return { _tag: 'Compared', mainCommit: main.commit, outcomes, failures: outcomes.filter(fails) } + }, + }) diff --git a/evals/ratstack-scorecard/src/model/dispatch.ts b/evals/ratstack-scorecard/src/model/dispatch.ts new file mode 100644 index 0000000..46811b6 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/dispatch.ts @@ -0,0 +1,38 @@ +import type { Cell, CitationCheck, Direction, Flag, Kind, MainBaseline, Ratchet, RowOutcome, Verdict } from './cell.ts' + +type ByTag = { [V in T as V['_tag']]: V } +export type Cases = { readonly [K in keyof M]: (value: M[K]) => R } + +const dispatch = (tag: K, value: M[K], cases: Cases): R => cases[tag](value) + +export const matchCell = (cell: Cell, cases: Cases, R>): R => + dispatch, Cell['_tag'], R>(cell._tag, cell, cases) + +export const matchCheck = (check: CitationCheck, cases: Cases, R>): R => + dispatch, CitationCheck['_tag'], R>(check._tag, check, cases) + +export const matchVerdict = (verdict: Verdict, cases: Cases, R>): R => + dispatch, Verdict['_tag'], R>(verdict._tag, verdict, cases) + +export const matchOutcome = (outcome: RowOutcome, cases: Cases, R>): R => + dispatch, RowOutcome['_tag'], R>(outcome._tag, outcome, cases) + +export const matchRatchet = (ratchet: Ratchet, cases: Cases, R>): R => + dispatch, Ratchet['_tag'], R>(ratchet._tag, ratchet, cases) + +export const matchBaseline = (baseline: MainBaseline, cases: Cases, R>): R => + dispatch, MainBaseline['_tag'], R>(baseline._tag, baseline, cases) + +export const matchFlag = (flag: Flag, cases: Cases, R>): R => + dispatch, Flag['_tag'], R>(flag._tag, flag, cases) + +export const matchDirection = (direction: Direction, cases: { readonly [K in Direction]: () => R }): R => + cases[direction]() + +export const matchKind = (kind: Kind, cases: { readonly [K in Kind]: () => R }): R => cases[kind]() + +export const firstRule = (rules: readonly (readonly [boolean, () => R])[], otherwise: () => R): R => + rules.reduceRight<() => R>( + (later, [guard, thunk]) => ({ true: thunk, false: later })[`${guard}` as const], + otherwise, + )() diff --git a/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts b/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts new file mode 100644 index 0000000..436932e --- /dev/null +++ b/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts @@ -0,0 +1,92 @@ +import { fc, test } from '@fast-check/vitest' +import { describe } from 'vitest' +import type { Cell, RowDefinition } from './cell.ts' +import { judgeRow } from './judge-row.workflow.ts' +import { cell, definition, runsFor } from './scorecard.arbitrary.ts' + +const measured = (runs: readonly number[]): Cell => ({ _tag: 'Measured', runs }) + +const negated = (c: Cell): Cell => c._tag === 'Measured' ? measured(c.runs.map((v) => -v)) : c + +const verifiedCitation: Cell = { + _tag: 'Unsupported', + citation: { + file: 'packages/core/src/join-interest-contract.ts', + lines: [22, 26], + text: 'Confirmation is not a seat', + }, + check: { _tag: 'Verified' }, +} + +const withRuns = (overrides: Partial = {}) => + definition(overrides).chain((d) => fc.tuple(fc.constant(d), runsFor(d))) + +describe('judgeRow', () => { + test.prop([withRuns(), fc.double({ min: 1e-3, max: 1e3, noNaN: true })])( + 'a starter strictly better than every rat-stack run is beaten, and swapping the sides is not', + ([d, runs], gap) => { + const spread = Math.max(...runs) - Math.min(...runs) + const better = { lower: -1, higher: 1 }[d.direction] + const ratstack = measured(runs) + const starter = measured(runs.map((v) => v + better * (spread + gap))) + return judgeRow({ definition: d, ratstack, starter })._tag === 'Beaten' && + judgeRow({ definition: d, ratstack: starter, starter: ratstack })._tag === 'NotBeaten' + }, + ) + + test.prop([ + definition({ kind: 'measurement' }), + fc.double({ min: -1e6, max: 1e6, noNaN: true }), + fc.array(fc.double({ min: 1e-3, max: 1e3, noNaN: true }), { minLength: 4, maxLength: 4 }), + ])("a starter whose every run only equals rat-stack's best run is not beaten", (d, best, gaps) => { + const worse = { lower: 1, higher: -1 }[d.direction] + const ratstack = measured([best, ...gaps.slice(0, d.runs - 1).map((gap) => best + worse * gap)]) + const starter = measured(Array.from({ length: d.runs }, () => best)) + return judgeRow({ definition: d, ratstack, starter })._tag === 'NotBeaten' + }) + + test.prop([withRuns()])( + 'identical run ranges are a tie', + ([d, runs]) => + judgeRow({ definition: d, ratstack: measured(runs), starter: measured([...runs].reverse()) })._tag === 'Tie', + ) + + test.prop([definition().chain((d) => fc.tuple(fc.constant(d), cell(d), cell(d)))])( + 'direction higher mirrors direction lower under negation', + ([d, ratstack, starter]) => { + const flipped = { ...d, direction: ({ lower: 'higher', higher: 'lower' } as const)[d.direction] } + return judgeRow({ definition: flipped, ratstack: negated(ratstack), starter: negated(starter) })._tag === + judgeRow({ definition: d, ratstack, starter })._tag + }, + ) + + test.prop([withRuns(), fc.integer({ min: 1, max: 4 }), fc.boolean()])( + 'a side with fewer runs than its definition is an instrument error', + ([d, runs], cut, cutStarter) => { + const short = measured(runs.slice(Math.min(cut, d.runs - 1))) + const full = measured(runs) + const [ratstack, starter] = cutStarter ? [full, short] : [short, full] + return judgeRow({ definition: d, ratstack, starter })._tag === 'InstrumentError' + }, + ) + + test.prop([withRuns({ kind: 'count' }), fc.double({ min: 1, max: 9, noNaN: true })])( + 'a starter count that does not reproduce is an instrument error, even against a verified citation', + ([d, runs], bump) => + judgeRow({ + definition: d, + ratstack: verifiedCitation, + starter: measured([...runs.slice(1), runs[0]! + bump]), + })._tag === 'InstrumentError', + ) + + test.prop([definition().chain((d) => fc.tuple(fc.constant(d), cell(d))), fc.string()])( + 'a contradicted rat-stack citation is an instrument error whatever the starter shows', + ([d, starter], found) => + judgeRow({ + definition: d, + ratstack: { ...verifiedCitation, check: { _tag: 'Contradicted', found } }, + starter, + })._tag === 'InstrumentError', + ) +}) diff --git a/evals/ratstack-scorecard/src/model/judge-row.workflow.ts b/evals/ratstack-scorecard/src/model/judge-row.workflow.ts new file mode 100644 index 0000000..406d9ca --- /dev/null +++ b/evals/ratstack-scorecard/src/model/judge-row.workflow.ts @@ -0,0 +1,89 @@ +import type { Cell, Citation, Kind, RowDefinition, Verdict } from './cell.ts' +import { firstRule, matchCell, matchCheck, matchKind } from './dispatch.ts' +import { smallerIsBetter } from './runs.ts' + +export interface JudgeRowInput { + readonly definition: RowDefinition + readonly ratstack: Cell + readonly starter: Cell +} + +const beaten: Verdict = { _tag: 'Beaten' } +const notBeaten: Verdict = { _tag: 'NotBeaten' } +const tie: Verdict = { _tag: 'Tie' } +const instrumentError = (error: string): Verdict => ({ _tag: 'InstrumentError', error }) + +const range = (runs: readonly number[]): string => `${Math.min(...runs)}..${Math.max(...runs)}` + +const countDisagrees = (kind: Kind, runs: readonly number[]): boolean => + matchKind(kind, { count: () => new Set(runs).size > 1, measurement: () => false }) + +const reproducible = ( + side: string, + definition: RowDefinition, + runs: readonly number[], + then: () => Verdict, +): Verdict => + firstRule([ + [ + runs.length !== definition.runs, + () => instrumentError(`${side} produced ${runs.length} of ${definition.runs} runs`), + ], + [ + countDisagrees(definition.kind, runs), + () => instrumentError(`${side} count runs disagree: ${runs.join(', ')}`), + ], + ], then) + +const compareRuns = (definition: RowDefinition, ratstack: readonly number[], starter: readonly number[]): Verdict => { + const r = smallerIsBetter(definition.direction, ratstack) + const s = smallerIsBetter(definition.direction, starter) + return firstRule([ + [range(r) === range(s), () => tie], + [Math.max(...s) < Math.min(...r), () => beaten], + ], () => notBeaten) +} + +const contradicted = (citation: Citation, found: string): Verdict => + instrumentError( + `rat-stack ${citation.file}:${citation.lines[0]}-${ + citation.lines[1] + } no longer contains "${citation.text}" (found "${found}")`, + ) + +const againstStarter = ( + input: JudgeRowInput, + cases: { readonly measured: (runs: readonly number[]) => Verdict; readonly unsupported: Verdict }, +): Verdict => + matchCell(input.starter, { + Measured: (cell) => reproducible('starter', input.definition, cell.runs, () => cases.measured(cell.runs)), + Unsupported: () => cases.unsupported, + Absent: () => notBeaten, + NoDeployment: () => notBeaten, + Unmeasurable: () => notBeaten, + NoSecret: () => notBeaten, + InstrumentError: (cell) => instrumentError(`starter: ${cell.error}`), + }) + +const againstNothing = (input: JudgeRowInput): Verdict => + againstStarter(input, { measured: () => notBeaten, unsupported: notBeaten }) + +export const judgeRow = (input: JudgeRowInput): Verdict => + matchCell(input.ratstack, { + Measured: (ratstack) => + reproducible('rat-stack', input.definition, ratstack.runs, () => + againstStarter(input, { + measured: (starter) => compareRuns(input.definition, ratstack.runs, starter), + unsupported: notBeaten, + })), + Unsupported: (ratstack) => + matchCheck(ratstack.check, { + Verified: () => againstStarter(input, { measured: () => beaten, unsupported: tie }), + Contradicted: (check) => contradicted(ratstack.citation, check.found), + }), + Absent: () => againstNothing(input), + NoDeployment: () => againstNothing(input), + Unmeasurable: () => againstNothing(input), + NoSecret: () => againstNothing(input), + InstrumentError: (ratstack) => instrumentError(`rat-stack: ${ratstack.error}`), + }) diff --git a/evals/ratstack-scorecard/src/model/runs.ts b/evals/ratstack-scorecard/src/model/runs.ts new file mode 100644 index 0000000..cb89727 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/runs.ts @@ -0,0 +1,12 @@ +import type { Direction } from './cell.ts' +import { matchDirection } from './dispatch.ts' + +export const smallerIsBetter = (direction: Direction, runs: readonly number[]): readonly number[] => + runs.map((value) => value * matchDirection(direction, { lower: () => 1, higher: () => -1 })) + +export const median = (runs: readonly number[]): number => { + const sorted = [...runs].sort((a, b) => a - b) + const lower = Math.ceil(sorted.length / 2) - 1 + const upper = Math.floor(sorted.length / 2) + return sorted.slice(lower, upper + 1).reduce((sum, value) => sum + value, 0) / (upper - lower + 1) +} diff --git a/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts b/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts new file mode 100644 index 0000000..b048343 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts @@ -0,0 +1,34 @@ +import { test } from '@fast-check/vitest' +import Ajv from 'ajv' +import { describe } from 'vitest' +import schema from '../../scorecard.schema.json' with { type: 'json' } +import { assembleScorecard } from './scorecard-document.ts' +import { assembleInput } from './scorecard.arbitrary.ts' + +const validate = new Ajv({ allErrors: true, strict: true }).compile(schema) + +const roundTrip = (value: unknown): unknown => JSON.parse(JSON.stringify(value)) + +describe('assembleScorecard', () => { + test.prop([assembleInput])( + 'every assembled document validates against scorecard.schema.json', + (input) => validate(roundTrip(assembleScorecard(input))), + ) + + test.prop([assembleInput])('a document without provenance.commit is refused by the schema', (input) => { + const { commit: _, ...provenance } = assembleScorecard(input).provenance + return !validate(roundTrip({ ...assembleScorecard(input), provenance })) + }) + + test.prop([assembleInput])( + 'a side with no cell for a row gets an instrument error naming the row', + (input) => + assembleScorecard(input).rows.every((row) => + (['ratstack', 'starter'] as const) + .filter((side) => !input.cells.some((cell) => cell.id === row.definition.id && cell.side === side)) + .every((side) => + row[side].cell._tag === 'InstrumentError' && row[side].cell.error.includes(row.definition.id) + ) + ), + ) +}) diff --git a/evals/ratstack-scorecard/src/model/scorecard-document.ts b/evals/ratstack-scorecard/src/model/scorecard-document.ts new file mode 100644 index 0000000..e7e8ee1 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/scorecard-document.ts @@ -0,0 +1,78 @@ +import type { + DocumentProvenance, + Flag, + MainBaseline, + MeasuredCell, + Row, + RowDefinition, + ScorecardDocument, + Side, +} from './cell.ts' +import { compareWithMain } from './compare-with-main.workflow.ts' +import { judgeRow } from './judge-row.workflow.ts' + +export interface DefinedRow { + readonly definition: RowDefinition + readonly hash: string +} + +export interface SideCell { + readonly id: string + readonly side: Side + readonly measured: MeasuredCell +} + +export interface RowFlag { + readonly id: string + readonly flag: Flag +} + +export interface AssembleInput { + readonly rows: readonly DefinedRow[] + readonly cells: readonly SideCell[] + readonly flags: readonly RowFlag[] + readonly provenance: DocumentProvenance + readonly main: MainBaseline +} + +const missingCell = (provenance: DocumentProvenance, id: string, side: Side): MeasuredCell => ({ + cell: { _tag: 'InstrumentError', error: `no ${side} cell was produced for ${id}` }, + provenance: { + side, + commit: { ratstack: provenance.ratstackCommit, starter: provenance.commit }[side], + instrumentHash: provenance.instrumentHash, + nixpkgsRev: provenance.nixpkgsRev, + runner: provenance.runner, + measuredAt: provenance.generatedAt, + tools: {}, + }, +}) + +const cellFor = (input: AssembleInput, id: string, side: Side): MeasuredCell => + input.cells + .filter((cell) => cell.id === id) + .filter((cell) => cell.side === side) + .reduce((_, cell) => cell.measured, missingCell(input.provenance, id, side)) + +const rowOf = (input: AssembleInput, defined: DefinedRow): Row => { + const ratstack = cellFor(input, defined.definition.id, 'ratstack') + const starter = cellFor(input, defined.definition.id, 'starter') + return { + definition: defined.definition, + definitionHash: defined.hash, + ratstack, + starter, + verdict: judgeRow({ definition: defined.definition, ratstack: ratstack.cell, starter: starter.cell }), + flags: input.flags.filter((flag) => flag.id === defined.definition.id).map((flag) => flag.flag), + } +} + +export const assembleScorecard = (input: AssembleInput): ScorecardDocument => { + const rows = input.rows.map((defined) => rowOf(input, defined)) + return { + schemaVersion: 1, + provenance: input.provenance, + rows, + ratchet: compareWithMain({ rows, main: input.main }), + } +} diff --git a/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts b/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts new file mode 100644 index 0000000..3655133 --- /dev/null +++ b/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts @@ -0,0 +1,139 @@ +import fc from 'fast-check' +import type { + Cell, + CellProvenance, + DocumentProvenance, + Flag, + MeasuredCell, + Row, + RowDefinition, + Side, + Verdict, +} from './cell.ts' +import type { AssembleInput } from './scorecard-document.ts' + +export const sha = fc.stringMatching(/^[0-9a-f]{40}$/) +const text = fc.string({ minLength: 1, maxLength: 400 }) +const value = fc.double({ min: -1e6, max: 1e6, noNaN: true, noDefaultInfinity: true }) + +export const definition = (overrides: Partial = {}): fc.Arbitrary => + fc.record({ + id: fc.stringMatching(/^M[0-9]{1,2}(\.[a-z0-9]{1,8})?$/), + metric: fc.stringMatching(/^M[0-9]{1,2}$/), + bin: text, + label: text, + unit: text, + direction: fc.constantFrom('lower' as const, 'higher' as const), + kind: fc.constantFrom('count' as const, 'measurement' as const), + runs: fc.constantFrom(3, 5), + family: fc.constantFrom( + 'static' as const, + 'cold-path' as const, + 'gate-mutation' as const, + 'running-stack' as const, + 'agent-surfaces' as const, + 'networked' as const, + ), + }).map((generated) => ({ ...generated, ...overrides })) + +export const runsFor = (row: RowDefinition): fc.Arbitrary => + ({ + count: () => value.map((v) => Array.from({ length: row.runs }, () => v)), + measurement: () => fc.array(value, { minLength: row.runs, maxLength: row.runs }), + })[row.kind]() + +export const cell = (row: RowDefinition): fc.Arbitrary => + fc.oneof( + runsFor(row).map((runs): Cell => ({ _tag: 'Measured', runs })), + text.map((reason): Cell => ({ _tag: 'Absent', reason })), + sha.map((s): Cell => ({ _tag: 'NoDeployment', sha: s })), + fc.record({ + file: text, + lines: fc.tuple(fc.integer({ min: 1, max: 999 }), fc.integer({ min: 1, max: 999 })), + text, + verified: fc.boolean(), + found: fc.string(), + }).map(({ file, lines, text, verified, found }): Cell => ({ + _tag: 'Unsupported', + citation: { file, lines, text }, + check: ({ true: { _tag: 'Verified' }, false: { _tag: 'Contradicted', found } } as const)[`${verified}`], + })), + text.map((error): Cell => ({ _tag: 'Unmeasurable', error })), + text.map((name): Cell => ({ _tag: 'NoSecret', name })), + text.map((error): Cell => ({ _tag: 'InstrumentError', error })), + ) + +export const cellProvenance = (side: Side): fc.Arbitrary => + fc.record({ + side: fc.constant(side), + commit: sha, + instrumentHash: sha, + nixpkgsRev: sha, + runner: text, + measuredAt: fc.date({ noInvalidDate: true }).map((d) => d.toISOString()), + tools: fc.dictionary(fc.string({ minLength: 1, maxLength: 20 }), fc.string({ maxLength: 20 })), + }) + +export const measuredCell = (row: RowDefinition, side: Side): fc.Arbitrary => + fc.record({ cell: cell(row), provenance: cellProvenance(side) }) + +export const documentProvenance: fc.Arbitrary = fc.record({ + commit: sha, + ratstackCommit: sha, + instrumentHash: sha, + nixpkgsRev: sha, + runner: text, + generatedAt: fc.date({ noInvalidDate: true }).map((d) => d.toISOString()), +}) + +export const verdict: fc.Arbitrary = fc.oneof( + fc.constant({ _tag: 'Beaten' }), + fc.constant({ _tag: 'NotBeaten' }), + fc.constant({ _tag: 'Tie' }), + text.map((error): Verdict => ({ _tag: 'InstrumentError', error })), +) + +const flag: fc.Arbitrary = fc.oneof( + fc.record({ _tag: fc.constant('LiveDiffersFromPin' as const), live: sha, pin: sha }), + fc.record({ _tag: fc.constant('ScannersDisagree' as const), primary: value, crossCheck: value }), +) + +export const row = (overrides: Partial = {}): fc.Arbitrary => + definition(overrides).chain((def) => + fc.record({ + definition: fc.constant(def), + definitionHash: sha, + ratstack: measuredCell(def, 'ratstack'), + starter: measuredCell(def, 'starter'), + verdict, + flags: fc.array(flag, { maxLength: 2 }), + }) + ) + +const uniqueDefinitions = fc.uniqueArray(definition(), { selector: (d) => d.id, maxLength: 64 }) + +export const assembleInput: fc.Arbitrary = uniqueDefinitions.chain((definitions) => + fc.record({ + rows: fc.tuple(...definitions.map((d) => sha.map((hash) => ({ definition: d, hash })))), + cells: fc.tuple( + ...definitions.flatMap((d) => + (['ratstack', 'starter'] as const).map((side) => + fc.option(measuredCell(d, side).map((measured) => ({ id: d.id, side, measured })), { nil: undefined }) + ) + ), + ).map((cells) => cells.filter((c) => c !== undefined)), + flags: fc.tuple( + ...definitions.map((d) => fc.array(flag, { maxLength: 2 }).map((fs) => fs.map((f) => ({ id: d.id, flag: f })))), + ) + .map((groups) => groups.flat()), + provenance: documentProvenance, + main: fc.oneof( + fc.constant({ _tag: 'Missing' as const }), + fc.record({ + _tag: fc.constant('Found' as const), + commit: sha, + rows: fc.array(row(), { maxLength: 8 }), + }), + ), + }) +) diff --git a/evals/ratstack-scorecard/src/model/summary-table.property.test.ts b/evals/ratstack-scorecard/src/model/summary-table.property.test.ts new file mode 100644 index 0000000..9ce385c --- /dev/null +++ b/evals/ratstack-scorecard/src/model/summary-table.property.test.ts @@ -0,0 +1,20 @@ +import { test } from '@fast-check/vitest' +import { describe } from 'vitest' +import { assembleScorecard } from './scorecard-document.ts' +import { assembleInput } from './scorecard.arbitrary.ts' +import { renderSummary } from './summary-table.ts' + +const stepSummaryLimit = 1024 * 1024 + +describe('renderSummary', () => { + test.prop([assembleInput])( + 'the summary of any registry-sized document fits the 1 MiB step summary limit', + (input) => new TextEncoder().encode(renderSummary(assembleScorecard(input))).byteLength < stepSummaryLimit, + ) + + test.prop([assembleInput])('the summary table has one line per row whatever the cell text holds', (input) => { + const doc = assembleScorecard(input) + const tableLines = renderSummary(doc).split('\n').filter((line) => line.startsWith('| ')).length + return tableLines === doc.rows.length + 2 + }) +}) diff --git a/evals/ratstack-scorecard/src/model/summary-table.ts b/evals/ratstack-scorecard/src/model/summary-table.ts new file mode 100644 index 0000000..5c159cd --- /dev/null +++ b/evals/ratstack-scorecard/src/model/summary-table.ts @@ -0,0 +1,100 @@ +import type { Cell, Flag, Kind, Row, RowOutcome, ScorecardDocument, Verdict } from './cell.ts' +import { matchCell, matchFlag, matchKind, matchOutcome, matchRatchet, matchVerdict } from './dispatch.ts' +import { median } from './runs.ts' + +const cellText = (text: string): string => text.replace(/\s+/g, ' ').replace(/\|/g, '\\|').slice(0, 160) + +const short = (sha: string): string => sha.slice(0, 7) + +const number = (value: number): string => String(Number(value.toFixed(3))) + +const measured = (kind: Kind, unit: string, runs: readonly number[]): string => + matchKind(kind, { + count: () => `${number(median(runs))} ${unit}`, + measurement: () => + `${number(median(runs))} ${unit} (${number(Math.min(...runs))}–${number(Math.max(...runs))}, n=${runs.length})`, + }) + +const showCell = (row: Row, cell: Cell): string => + matchCell(cell, { + Measured: (c) => measured(row.definition.kind, row.definition.unit, c.runs), + Absent: (c) => `absent: ${c.reason}`, + NoDeployment: (c) => `absent: no deployment for ${short(c.sha)}`, + Unsupported: (c) => `unsupported: ${c.citation.file}:${c.citation.lines[0]}-${c.citation.lines[1]}`, + Unmeasurable: (c) => `unmeasurable: ${c.error}`, + NoSecret: (c) => `no secret: ${c.name}`, + InstrumentError: (c) => `instrument error: ${c.error}`, + }) + +const showVerdict = (verdict: Verdict): string => + matchVerdict(verdict, { + Beaten: () => '**beaten**', + NotBeaten: () => 'not beaten', + Tie: () => 'tie', + InstrumentError: () => '**instrument error**', + }) + +const showFlag = (flag: Flag): string => + matchFlag(flag, { + LiveDiffersFromPin: (f) => `live ${short(f.live)}≠pin`, + ScannersDisagree: (f) => `scanners disagree (${f.primary} vs ${f.crossCheck})`, + }) + +const showOutcome = (outcome: RowOutcome): string => + matchOutcome(outcome, { + Held: () => 'held', + New: () => 'new', + ReBaselined: () => 're-baselined', + Neutral: (o) => `neutral (${o.cause})`, + InstrumentError: (o) => `**failed**: ${o.error}`, + LostBeaten: (o) => `**failed**: no longer beaten (rat-stack ${o.ratstack}, starter ${o.starter})`, + Regressed: (o) => `**failed**: regressed from ${o.main} to ${o.pr}`, + }) + +const outcomesOf = (doc: ScorecardDocument): readonly RowOutcome[] => + matchRatchet(doc.ratchet, { FirstBaseline: (r) => r.outcomes, Compared: (r) => r.outcomes }) + +const failuresOf = (doc: ScorecardDocument): readonly RowOutcome[] => + matchRatchet(doc.ratchet, { FirstBaseline: (r) => r.failures, Compared: (r) => r.failures }) + +const ratchetLine = (doc: ScorecardDocument): string => + matchRatchet(doc.ratchet, { + FirstBaseline: (r) => `first baseline (no \`main\` artifact), ${r.failures.length} failing rows`, + Compared: (r) => `compared with \`main\` at \`${short(r.mainCommit)}\`, ${r.failures.length} failing rows`, + }) + +const tableRow = (row: Row, outcomes: readonly RowOutcome[]): string => + `| ${ + [ + row.definition.id, + row.definition.bin, + row.definition.label, + showCell(row, row.ratstack.cell), + showCell(row, row.starter.cell), + showVerdict(row.verdict), + outcomes.filter((o) => o.id === row.definition.id).map(showOutcome).join(', '), + row.flags.map(showFlag).join(', '), + ].map(cellText).join(' | ') + } |` + +export const renderSummary = (doc: ScorecardDocument): string => + [ + '## rat-stack scorecard', + '', + `Starter \`${short(doc.provenance.commit)}\` against rat-stack \`${ + short(doc.provenance.ratstackCommit) + }\`, instrument \`${short(doc.provenance.instrumentHash)}\`, nixpkgs \`${ + short(doc.provenance.nixpkgsRev) + }\`, runner \`${cellText(doc.provenance.runner)}\`, ${doc.provenance.generatedAt}.`, + '', + `**Ratchet:** ${ratchetLine(doc)}.`, + '', + ...failuresOf(doc).map((failure) => `- ${failure.id}: ${cellText(showOutcome(failure))}`), + '', + `Beaten: ${doc.rows.filter((row) => row.verdict._tag === 'Beaten').length} of ${doc.rows.length} rows.`, + '', + '| Row | Bin | Metric | rat-stack | starter | Verdict | Ratchet | Flags |', + '| --- | --- | --- | --- | --- | --- | --- | --- |', + ...doc.rows.map((row) => tableRow(row, outcomesOf(doc))), + '', + ].join('\n') diff --git a/evals/ratstack-scorecard/vitest.config.ts b/evals/ratstack-scorecard/vitest.config.ts new file mode 100644 index 0000000..5c50db5 --- /dev/null +++ b/evals/ratstack-scorecard/vitest.config.ts @@ -0,0 +1,10 @@ +import { defineConfig } from 'vitest/config' + +export default defineConfig({ + test: { + projects: [ + { test: { name: 'model', environment: 'node', include: ['src/**/*.test.ts'] } }, + { test: { name: 'journeys', environment: 'node', include: ['journeys/**/*.journey.test.ts'] } }, + ], + }, +}) From d909baf9b9eda8fd019110b0e4d0fd0ee2fff891 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Tue, 6 Oct 2026 01:33:53 +0000 Subject: [PATCH 3/7] build(repo): install the scorecard tools offline from a Nix-built store The instrument gets its own flake: nixpkgs 494ce7f (pnpm 12.9.0, per NixOS/nixpkgs#566850), the pnpm-release-management launcher, and a fixed-output pnpm store built from pnpm-lock.yaml. Installs run in the launcher with no allowed hosts, so nothing reaches the registry. Drops the storeDir override, which empties the fetcher output Review fix for #44 from Kiro --- evals/ratstack-scorecard/README.md | 7 +- evals/ratstack-scorecard/flake.lock | 94 ++++++++++++++++++++ evals/ratstack-scorecard/flake.nix | 42 +++++++++ evals/ratstack-scorecard/pnpm-workspace.yaml | 1 - 4 files changed, 139 insertions(+), 5 deletions(-) create mode 100644 evals/ratstack-scorecard/flake.lock create mode 100644 evals/ratstack-scorecard/flake.nix diff --git a/evals/ratstack-scorecard/README.md b/evals/ratstack-scorecard/README.md index 7c75285..691454a 100644 --- a/evals/ratstack-scorecard/README.md +++ b/evals/ratstack-scorecard/README.md @@ -19,13 +19,12 @@ The scorecard measures the starter against [rat-stack](https://github.com/joelho ## Running it -Everything that loads third-party code runs inside the sandbox launcher from `systemfsoftware/pnpm-release-management` (`packages..sandbox`), with this directory as the sandbox project: +Everything that loads third-party code runs inside the sandbox launcher from `systemfsoftware/pnpm-release-management` (`packages..sandbox`), with this directory as the sandbox project. The instrument's own flake (`flake.nix`) pins nixpkgs (pnpm 12.9.0, Node 24) and the launcher, and builds the tools' pnpm store from `pnpm-lock.yaml` as a fixed-output derivation (`tools-store`). The install is offline from that store; the sandbox gets no network at all. ```sh cd evals/ratstack-scorecard -export SANDBOX_PROJECT=$PWD -sandbox --allow-host registry.npmjs.org -- pnpm install --frozen-lockfile -sandbox -- pnpm vitest run --project model +nix develop --command sh -c 'SANDBOX_PROJECT=$PWD sandbox --pnpm-store "$SANDBOX_PNPM_STORE" -- pnpm install --frozen-lockfile' +nix develop --command sh -c 'SANDBOX_PROJECT=$PWD sandbox -- pnpm vitest run --project model' ``` The decision modules (`src/model/*.workflow.ts`) and the orchestrator import only Deno APIs, `node:` builtins and each other, so `deno check src/` type-checks them without any third-party code. diff --git a/evals/ratstack-scorecard/flake.lock b/evals/ratstack-scorecard/flake.lock new file mode 100644 index 0000000..c464b2c --- /dev/null +++ b/evals/ratstack-scorecard/flake.lock @@ -0,0 +1,94 @@ +{ + "nodes": { + "comment-checker": { + "inputs": { + "nixpkgs": [ + "pnpm-release-management", + "nixpkgs" + ], + "rust-overlay": "rust-overlay" + }, + "locked": { + "lastModified": 1788142907, + "narHash": "sha256-arhhsgBbDTOTtnG9gphKARLXkCj3emOZUm3sP0a0jnI=", + "owner": "systemfsoftware", + "repo": "comment-checker", + "rev": "60e60bbedb2686c807a9ca64a31f7bba4e75b97b", + "type": "github" + }, + "original": { + "owner": "systemfsoftware", + "repo": "comment-checker", + "type": "github" + } + }, + "nixpkgs": { + "locked": { + "lastModified": 1791179034, + "narHash": "sha256-Ni0LBydzaCi8oek12r4mVreuFHqYlM1eTD6SfiENKfw=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "494ce7fd23ff6a5dff39e1fb11e9b6f2ac74bf25", + "type": "github" + }, + "original": { + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "494ce7fd23ff6a5dff39e1fb11e9b6f2ac74bf25", + "type": "github" + } + }, + "pnpm-release-management": { + "inputs": { + "comment-checker": "comment-checker", + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1791237748, + "narHash": "sha256-IWdNptNMXzjs2yG3DRQJKUkRcOmVvyDQ0x0dN2maEVU=", + "owner": "systemfsoftware", + "repo": "pnpm-release-management", + "rev": "180122866dd537fa728b5563fb1820fbd2af88cc", + "type": "github" + }, + "original": { + "owner": "systemfsoftware", + "repo": "pnpm-release-management", + "rev": "180122866dd537fa728b5563fb1820fbd2af88cc", + "type": "github" + } + }, + "root": { + "inputs": { + "nixpkgs": "nixpkgs", + "pnpm-release-management": "pnpm-release-management" + } + }, + "rust-overlay": { + "inputs": { + "nixpkgs": [ + "pnpm-release-management", + "comment-checker", + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1787367552, + "narHash": "sha256-YT4Fs2k7bi+7YzuLt93EtIRgjpwHK5ZfsQEIh5dEQSk=", + "owner": "oxalica", + "repo": "rust-overlay", + "rev": "fd2ebb9cc4323d0c5a1336138dab5c3c5a5d8bd9", + "type": "github" + }, + "original": { + "owner": "oxalica", + "repo": "rust-overlay", + "type": "github" + } + } + }, + "root": "root", + "version": 7 +} diff --git a/evals/ratstack-scorecard/flake.nix b/evals/ratstack-scorecard/flake.nix new file mode 100644 index 0000000..f54425e --- /dev/null +++ b/evals/ratstack-scorecard/flake.nix @@ -0,0 +1,42 @@ +{ + description = "rat-stack scorecard: the verifier-owned instrument that measures the starter against rat-stack"; + + inputs = { + nixpkgs.url = "github:NixOS/nixpkgs/494ce7fd23ff6a5dff39e1fb11e9b6f2ac74bf25"; + pnpm-release-management = { + url = "github:systemfsoftware/pnpm-release-management/180122866dd537fa728b5563fb1820fbd2af88cc"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + }; + + outputs = { self, nixpkgs, pnpm-release-management }: + let + systems = [ "x86_64-linux" "aarch64-linux" "aarch64-darwin" ]; + forEachSystem = fn: nixpkgs.lib.genAttrs systems (system: fn nixpkgs.legacyPackages.${system}); + in + { + packages = forEachSystem (pkgs: + let + system = pkgs.stdenv.hostPlatform.system; + sandbox = pnpm-release-management.packages.${system}.sandbox; + tools-store = (pnpm-release-management.lib.mkPnpmWorkspacePackages { + inherit pkgs; + src = self; + pname = "ratstack-scorecard"; + pnpm = pkgs.pnpm_12; + hash = "sha256-TylxLEQflTlKx6QDk9mFlBOEInUvvBBeOVFUYCURmYo="; + }).pnpm-store; + in + { + inherit sandbox tools-store; + }); + + devShells = forEachSystem (pkgs: + let system = pkgs.stdenv.hostPlatform.system; in { + default = pkgs.mkShell { + packages = [ pkgs.nodejs_24 pkgs.pnpm_12 pkgs.deno self.packages.${system}.sandbox ]; + SANDBOX_PNPM_STORE = self.packages.${system}.tools-store; + }; + }); + }; +} diff --git a/evals/ratstack-scorecard/pnpm-workspace.yaml b/evals/ratstack-scorecard/pnpm-workspace.yaml index d728e2b..d05a7e7 100644 --- a/evals/ratstack-scorecard/pnpm-workspace.yaml +++ b/evals/ratstack-scorecard/pnpm-workspace.yaml @@ -1,3 +1,2 @@ packages: - . -storeDir: node_modules/.pnpm-store From f8914272e28db49be4d010b90daf486a8243e7a4 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Tue, 6 Oct 2026 04:28:19 +0000 Subject: [PATCH 4/7] test(repo): run the scorecard tests as one vitest project Review fix #1/#3 (CONST-T12): no project split by folder or suffix. A test becomes a journey by importing the journey fixture, added with the first journey --- evals/ratstack-scorecard/README.md | 2 +- evals/ratstack-scorecard/vitest.config.ts | 6 ++---- 2 files changed, 3 insertions(+), 5 deletions(-) diff --git a/evals/ratstack-scorecard/README.md b/evals/ratstack-scorecard/README.md index 691454a..e701639 100644 --- a/evals/ratstack-scorecard/README.md +++ b/evals/ratstack-scorecard/README.md @@ -24,7 +24,7 @@ Everything that loads third-party code runs inside the sandbox launcher from `sy ```sh cd evals/ratstack-scorecard nix develop --command sh -c 'SANDBOX_PROJECT=$PWD sandbox --pnpm-store "$SANDBOX_PNPM_STORE" -- pnpm install --frozen-lockfile' -nix develop --command sh -c 'SANDBOX_PROJECT=$PWD sandbox -- pnpm vitest run --project model' +nix develop --command sh -c 'SANDBOX_PROJECT=$PWD sandbox -- pnpm vitest run' ``` The decision modules (`src/model/*.workflow.ts`) and the orchestrator import only Deno APIs, `node:` builtins and each other, so `deno check src/` type-checks them without any third-party code. diff --git a/evals/ratstack-scorecard/vitest.config.ts b/evals/ratstack-scorecard/vitest.config.ts index 5c50db5..bf7e742 100644 --- a/evals/ratstack-scorecard/vitest.config.ts +++ b/evals/ratstack-scorecard/vitest.config.ts @@ -2,9 +2,7 @@ import { defineConfig } from 'vitest/config' export default defineConfig({ test: { - projects: [ - { test: { name: 'model', environment: 'node', include: ['src/**/*.test.ts'] } }, - { test: { name: 'journeys', environment: 'node', include: ['journeys/**/*.journey.test.ts'] } }, - ], + environment: 'node', + maxWorkers: 4, }, }) From 14683927f4a92005c49bd63216689e16ba3b72b4 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Tue, 6 Oct 2026 04:28:20 +0000 Subject: [PATCH 5/7] fix(repo): require an absolute bar before an unsupported rat-stack row counts as beaten Review fix #4: every row declares ratstackSupport. Required rows turn an unsupported rat-stack cell into an instrument error; MayBeUnsupported rows (M12, bar 0) are beaten only when the starter's worst run meets the bar. Review fix #18: assembled cells are keyed by row and side, so a duplicate cell cannot reach assembly and cellFor no longer folds --- ...10-05-2151-feat-ratstack-scorecard-plan.md | 25 +++++++------ .../ratstack-scorecard/scorecard.schema.json | 20 +++++++++- .../src/metrics/registry.ts | 37 +++++++++++++++++++ .../src/model/acceptance-examples.test.ts | 26 ++++++++++++- evals/ratstack-scorecard/src/model/cell.ts | 5 +++ .../ratstack-scorecard/src/model/dispatch.ts | 16 +++++++- .../model/judge-row.workflow.property.test.ts | 28 ++++++++++++++ .../src/model/judge-row.workflow.ts | 19 +++++++++- .../model/scorecard-document.property.test.ts | 2 +- .../src/model/scorecard-document.ts | 9 ++--- .../src/model/scorecard.arbitrary.ts | 15 +++++++- 11 files changed, 177 insertions(+), 25 deletions(-) diff --git a/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md b/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md index 60db791..9aa43a0 100644 --- a/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md +++ b/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md @@ -205,7 +205,11 @@ CellStatus = Measured{runs[]} | Absent{reason} | Unsupported{citation} | Unmeasu | NoSecret{name} | InstrumentError{error} Verdict = match (ratstack, starter): (Measured, Measured) -> compare per KTD1 -> Beaten | NotBeaten | Tie - (Unsupported, Measured) -> Beaten when the citation re-verified at the pin, else InstrumentError + (Unsupported, Measured) -> citation contradicted at the pin: InstrumentError + row declares Required: InstrumentError + row declares MayBeUnsupported{bar}: Beaten only when the starter's worst + run meets bar in the row's direction (M12: 0 oversold seats on every run), + else NotBeaten; a non-reproducing starter count stays InstrumentError (_, Absent | NoSecret) -> NotBeaten (neutral for the ratchet when NoSecret or fork-Absent) (InstrumentError, _) | (_, InstrumentError) -> InstrumentError (Unmeasurable, _) -> NotBeaten, surfaced for Kiro @@ -274,7 +278,7 @@ U1 → U2 → U3 → U4 → U5 → U6 → U7 → U8 → U9. Each is one PR layer - AE2: unequal count runs give `instrument-error`. - Swapping sides of a strictly separated `measurement` turns `beaten` into `not-beaten`, and identical ranges are always `tie`. - Direction `higher` mirrors direction `lower` under negation. - - AE3: an `Unsupported` cell whose citation check failed gives `instrument-error`, and one that passed gives `beaten`. + - AE3: an `Unsupported` cell whose citation check failed gives `instrument-error`. One that passed gives `beaten` only on a row that declares `MayBeUnsupported` with a bar the starter's worst run meets (M12 `[0,0,0]` beaten, `[5,5,5]` not beaten); a row that declares `Required` gives `instrument-error`. - AE5: a PR median inside `main`'s run range passes the ratchet, and a median just past `main`'s worst run fails, naming the metric. - AE4: `beaten` on `main` and `not-beaten` on the PR fails the ratchet, and the failure names the metric and the cell status that caused it (for example a rat-stack `Unmeasurable` cell, or a changed live commit). - Present on `main` and `absent` on the PR fails, while `NoSecret`, fork `Absent`, or a changed metric-definition hash (`re-baselined`) is neutral. @@ -435,15 +439,14 @@ U1 → U2 → U3 → U4 → U5 → U6 → U7 → U8 → U9. Each is one PR layer ## Verification Contract -| Gate | Command | Applies to | -| --------------------- | ------------------------------------------------------------------------------------- | ---------------------------------------------- | -| Format (START-1) | `pnpm format:check` (dprint covers `evals/**`) | every unit | -| Instrument properties | `pnpm vitest run --project model` under the launcher, from `evals/ratstack-scorecard` | every unit | -| Journeys J1-J3 | `pnpm vitest run --project journeys` under the launcher | U2 (J2), U3 (J3), U4 (J1) and every later unit | -| Family run | `nix run ./evals/ratstack-scorecard#scorecard -- measure --family ` | U2, U5-U9 | -| Workflow lint | `actionlint` | U3 | -| Repo gate (START-4) | `pnpm check:ci` | every unit before its PR | -| Sabotage | break one decision or one probe, show the test red, revert | every unit; recorded in the PR body | +| Gate | Command | Applies to | +| ------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------- | +| Format (START-1) | `pnpm format:check` (dprint covers `evals/**`) | every unit | +| Instrument tests | `pnpm vitest run` under the launcher, from `evals/ratstack-scorecard`: one project; a test is a journey because it imports the journey fixture (`journeys/fixture.ts`), which carries the sandbox, subprocess, timeout and isolation | every unit | +| Family run | `nix run ./evals/ratstack-scorecard#scorecard -- measure --family ` | U2, U5-U9 | +| Workflow lint | `actionlint` | U3 | +| Repo gate (START-4) | `pnpm check:ci` | every unit before its PR | +| Sabotage | break one decision or one probe, show the test red, revert | every unit; recorded in the PR body | Mutation testing never runs locally; it runs only in the scorecard workflow's `mutation` job on push to `main` (KTD15). PR bodies carry commands, outputs and sabotage evidence, and the verifier runs every check in this session (VER1). diff --git a/evals/ratstack-scorecard/scorecard.schema.json b/evals/ratstack-scorecard/scorecard.schema.json index 117c072..fdbdcab 100644 --- a/evals/ratstack-scorecard/scorecard.schema.json +++ b/evals/ratstack-scorecard/scorecard.schema.json @@ -193,7 +193,7 @@ "rowDefinition": { "type": "object", "additionalProperties": false, - "required": ["id", "metric", "bin", "label", "unit", "direction", "kind", "runs", "family"], + "required": ["id", "metric", "bin", "label", "unit", "direction", "kind", "runs", "family", "ratstackSupport"], "properties": { "id": { "type": "string", "pattern": "^M[0-9]+(\\.[a-z0-9-]+)?$" }, "metric": { "type": "string", "pattern": "^M[0-9]+$" }, @@ -203,7 +203,23 @@ "direction": { "enum": ["lower", "higher"] }, "kind": { "enum": ["count", "measurement"] }, "runs": { "type": "integer", "minimum": 1 }, - "family": { "enum": ["static", "cold-path", "gate-mutation", "running-stack", "agent-surfaces", "networked"] } + "family": { "enum": ["static", "cold-path", "gate-mutation", "running-stack", "agent-surfaces", "networked"] }, + "ratstackSupport": { + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["_tag"], + "properties": { "_tag": { "const": "Required" } } + }, + { + "type": "object", + "additionalProperties": false, + "required": ["_tag", "bar"], + "properties": { "_tag": { "const": "MayBeUnsupported" }, "bar": { "type": "number" } } + } + ] + } } }, "row": { diff --git a/evals/ratstack-scorecard/src/metrics/registry.ts b/evals/ratstack-scorecard/src/metrics/registry.ts index c3dfbea..d1e364b 100644 --- a/evals/ratstack-scorecard/src/metrics/registry.ts +++ b/evals/ratstack-scorecard/src/metrics/registry.ts @@ -3,6 +3,7 @@ import type { Family, RowDefinition } from '../model/cell.ts' const count3 = { kind: 'count', runs: 3 } as const const measured5 = { kind: 'measurement', runs: 5 } as const const measured3 = { kind: 'measurement', runs: 3 } as const +const required = { ratstackSupport: { _tag: 'Required' } } as const export const familyTimeoutMinutes: Readonly> = { 'static': 20, @@ -23,6 +24,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'networked', + ...required, }, { id: 'M1.level', @@ -33,6 +35,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'networked', + ...required, }, { id: 'M2', @@ -43,6 +46,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'networked', + ...required, }, { id: 'M3.performance', @@ -53,6 +57,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...measured5, family: 'networked', + ...required, }, { id: 'M3.lcp', @@ -63,6 +68,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'networked', + ...required, }, { id: 'M3.cls', @@ -73,6 +79,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'networked', + ...required, }, { id: 'M3.tbt', @@ -83,6 +90,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'networked', + ...required, }, { id: 'M3.bytes', @@ -93,6 +101,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'networked', + ...required, }, { id: 'M3.a11y', @@ -103,6 +112,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'networked', + ...required, }, { id: 'M4', @@ -113,6 +123,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'agent-surfaces', + ...required, }, { id: 'M5', @@ -123,6 +134,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'agent-surfaces', + ...required, }, { id: 'M6', @@ -133,6 +145,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'cold-path', + ...required, }, { id: 'M7', @@ -143,6 +156,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'cold-path', + ...required, }, { id: 'M8', @@ -153,6 +167,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'agent-surfaces', + ...required, }, { id: 'M9', @@ -163,6 +178,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'agent-surfaces', + ...required, }, { id: 'M10.seconds', @@ -173,6 +189,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured5, family: 'agent-surfaces', + ...required, }, { id: 'M10.correct', @@ -183,6 +200,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...measured5, family: 'agent-surfaces', + ...required, }, { id: 'M11', @@ -193,6 +211,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'running-stack', + ...required, }, { id: 'M12', @@ -203,6 +222,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'running-stack', + ratstackSupport: { _tag: 'MayBeUnsupported', bar: 0 }, }, { id: 'M13', @@ -213,6 +233,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'gate-mutation', + ...required, }, { id: 'M14', @@ -223,6 +244,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'running-stack', + ...required, }, { id: 'M15', @@ -233,6 +255,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'agent-surfaces', + ...required, }, { id: 'M16', @@ -243,6 +266,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'agent-surfaces', + ...required, }, { id: 'M17', @@ -253,6 +277,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'running-stack', + ...required, }, { id: 'M18', @@ -263,6 +288,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'agent-surfaces', + ...required, }, { id: 'M19', @@ -273,6 +299,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'running-stack', + ...required, }, { id: 'M20', @@ -283,6 +310,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'gate-mutation', + ...required, }, { id: 'M21', @@ -293,6 +321,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'gate-mutation', + ...required, }, { id: 'M22', @@ -303,6 +332,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'gate-mutation', + ...required, }, { id: 'M23', @@ -313,6 +343,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'static', + ...required, }, { id: 'M24', @@ -323,6 +354,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'gate-mutation', + ...required, }, { id: 'M25', @@ -333,6 +365,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'gate-mutation', + ...required, }, { id: 'M26', @@ -343,6 +376,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...measured3, family: 'cold-path', + ...required, }, { id: 'M27', @@ -353,6 +387,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'cold-path', + ...required, }, { id: 'M28', @@ -363,6 +398,7 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'lower', ...count3, family: 'static', + ...required, }, { id: 'M29', @@ -373,5 +409,6 @@ export const rowDefinitions: readonly RowDefinition[] = [ direction: 'higher', ...count3, family: 'agent-surfaces', + ...required, }, ] diff --git a/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts b/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts index 751f8fc..29877f6 100644 --- a/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts +++ b/evals/ratstack-scorecard/src/model/acceptance-examples.test.ts @@ -13,10 +13,18 @@ const lcp: RowDefinition = { kind: 'measurement', runs: 5, family: 'networked', + ratstackSupport: { _tag: 'Required' }, } const debt: RowDefinition = { ...lcp, id: 'M23', metric: 'M23', kind: 'count', runs: 3, family: 'static' } const passwords: RowDefinition = { ...debt, id: 'M14', metric: 'M14', family: 'running-stack' } const cli: RowDefinition = { ...lcp, id: 'M9', metric: 'M9', family: 'agent-surfaces' } +const oversold: RowDefinition = { + ...debt, + id: 'M12', + metric: 'M12', + family: 'running-stack', + ratstackSupport: { _tag: 'MayBeUnsupported', bar: 0 }, +} const measured = (runs: readonly number[]): Cell => ({ _tag: 'Measured', runs }) @@ -69,11 +77,25 @@ describe('plan acceptance examples', () => { citation: seats, check: { _tag: 'Contradicted', found: 'export const seats = confirmations' }, } - expect(judgeRow({ definition: debt, ratstack: contradicted, starter })._tag).toBe('InstrumentError') - expect(judgeRow({ definition: debt, ratstack: { ...contradicted, check: { _tag: 'Verified' } }, starter })._tag) + expect(judgeRow({ definition: oversold, ratstack: contradicted, starter })._tag).toBe('InstrumentError') + expect(judgeRow({ definition: oversold, ratstack: { ...contradicted, check: { _tag: 'Verified' } }, starter })._tag) .toBe('Beaten') }) + test('M12: rat-stack unsupported is beaten only by a starter that oversells nothing on every run', () => { + const ratstack: Cell = { _tag: 'Unsupported', citation: seats, check: { _tag: 'Verified' } } + const verdict = (runs: readonly number[]) => + judgeRow({ definition: oversold, ratstack, starter: measured(runs) })._tag + expect(verdict([5, 5, 5])).toBe('NotBeaten') + expect(verdict([0, 0, 0])).toBe('Beaten') + expect(verdict([0, 1, 0])).toBe('InstrumentError') + }) + + test('a row that requires a rat-stack measurement turns an unsupported rat-stack cell into an instrument error', () => { + const ratstack: Cell = { _tag: 'Unsupported', citation: seats, check: { _tag: 'Verified' } } + expect(judgeRow({ definition: debt, ratstack, starter: measured([0, 0, 0]) })._tag).toBe('InstrumentError') + }) + test('AE4: main beats rat-stack on M14 and a PR re-enabling a password route fails, naming M14', () => { const main = row(passwords, measured([1, 1, 1]), measured([0, 0, 0])) const pr = row(passwords, measured([1, 1, 1]), measured([1, 1, 1])) diff --git a/evals/ratstack-scorecard/src/model/cell.ts b/evals/ratstack-scorecard/src/model/cell.ts index aaec3a6..b6921d8 100644 --- a/evals/ratstack-scorecard/src/model/cell.ts +++ b/evals/ratstack-scorecard/src/model/cell.ts @@ -20,8 +20,13 @@ export interface RowDefinition { readonly kind: Kind readonly runs: number readonly family: Family + readonly ratstackSupport: RatstackSupport } +export type RatstackSupport = + | { readonly _tag: 'Required' } + | { readonly _tag: 'MayBeUnsupported'; readonly bar: number } + export interface Citation { readonly file: string readonly lines: readonly [number, number] diff --git a/evals/ratstack-scorecard/src/model/dispatch.ts b/evals/ratstack-scorecard/src/model/dispatch.ts index 46811b6..26912f5 100644 --- a/evals/ratstack-scorecard/src/model/dispatch.ts +++ b/evals/ratstack-scorecard/src/model/dispatch.ts @@ -1,4 +1,15 @@ -import type { Cell, CitationCheck, Direction, Flag, Kind, MainBaseline, Ratchet, RowOutcome, Verdict } from './cell.ts' +import type { + Cell, + CitationCheck, + Direction, + Flag, + Kind, + MainBaseline, + Ratchet, + RatstackSupport, + RowOutcome, + Verdict, +} from './cell.ts' type ByTag = { [V in T as V['_tag']]: V } export type Cases = { readonly [K in keyof M]: (value: M[K]) => R } @@ -26,6 +37,9 @@ export const matchBaseline = (baseline: MainBaseline, cases: Cases(flag: Flag, cases: Cases, R>): R => dispatch, Flag['_tag'], R>(flag._tag, flag, cases) +export const matchSupport = (support: RatstackSupport, cases: Cases, R>): R => + dispatch, RatstackSupport['_tag'], R>(support._tag, support, cases) + export const matchDirection = (direction: Direction, cases: { readonly [K in Direction]: () => R }): R => cases[direction]() diff --git a/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts b/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts index 436932e..cbe7754 100644 --- a/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts +++ b/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts @@ -34,6 +34,34 @@ describe('judgeRow', () => { }, ) + test.prop([ + definition({ kind: 'measurement' }), + fc.double({ min: -1e6, max: 1e6, noNaN: true }), + fc.double({ min: 1e-3, max: 1e3, noNaN: true }), + ])( + "against an unsupported rat-stack, the starter is beaten exactly when its worst run meets the row's bar", + (base, bar, gap) => { + const d = { ...base, ratstackSupport: { _tag: 'MayBeUnsupported', bar } } as const + const better = { lower: -1, higher: 1 }[d.direction] + const verdict = (worst: number) => + judgeRow({ + definition: d, + ratstack: verifiedCitation, + starter: measured([...Array(d.runs - 1).fill(bar + better * gap), worst]), + }) + ._tag + return verdict(bar) === 'Beaten' && verdict(bar + better * gap) === 'Beaten' && + verdict(bar - better * gap) === 'NotBeaten' + }, + ) + + test.prop([definition({ ratstackSupport: { _tag: 'Required' } }), fc.double({ min: -1e6, max: 1e6, noNaN: true })])( + 'a row requiring a rat-stack measurement never counts an unsupported rat-stack as beaten', + (d, v) => + judgeRow({ definition: d, ratstack: verifiedCitation, starter: measured(Array(d.runs).fill(v)) })._tag === + 'InstrumentError', + ) + test.prop([ definition({ kind: 'measurement' }), fc.double({ min: -1e6, max: 1e6, noNaN: true }), diff --git a/evals/ratstack-scorecard/src/model/judge-row.workflow.ts b/evals/ratstack-scorecard/src/model/judge-row.workflow.ts index 406d9ca..2bcf7c3 100644 --- a/evals/ratstack-scorecard/src/model/judge-row.workflow.ts +++ b/evals/ratstack-scorecard/src/model/judge-row.workflow.ts @@ -1,5 +1,5 @@ import type { Cell, Citation, Kind, RowDefinition, Verdict } from './cell.ts' -import { firstRule, matchCell, matchCheck, matchKind } from './dispatch.ts' +import { firstRule, matchCell, matchCheck, matchKind, matchSupport } from './dispatch.ts' import { smallerIsBetter } from './runs.ts' export interface JudgeRowInput { @@ -65,6 +65,21 @@ const againstStarter = ( InstrumentError: (cell) => instrumentError(`starter: ${cell.error}`), }) +const meetsBar = (definition: RowDefinition, bar: number, runs: readonly number[]): boolean => + Math.max(...smallerIsBetter(definition.direction, runs)) <= Math.min(...smallerIsBetter(definition.direction, [bar])) + +const againstBar = (input: JudgeRowInput): Verdict => + matchSupport(input.definition.ratstackSupport, { + Required: () => + instrumentError(`rat-stack is unsupported on ${input.definition.id}, which requires a rat-stack measurement`), + MayBeUnsupported: ({ bar }) => + againstStarter(input, { + measured: (runs) => + firstRule([[meetsBar(input.definition, bar, runs), () => beaten]], () => notBeaten), + unsupported: tie, + }), + }) + const againstNothing = (input: JudgeRowInput): Verdict => againstStarter(input, { measured: () => notBeaten, unsupported: notBeaten }) @@ -78,7 +93,7 @@ export const judgeRow = (input: JudgeRowInput): Verdict => })), Unsupported: (ratstack) => matchCheck(ratstack.check, { - Verified: () => againstStarter(input, { measured: () => beaten, unsupported: tie }), + Verified: () => againstBar(input), Contradicted: (check) => contradicted(ratstack.citation, check.found), }), Absent: () => againstNothing(input), diff --git a/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts b/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts index b048343..b0b3ce8 100644 --- a/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts +++ b/evals/ratstack-scorecard/src/model/scorecard-document.property.test.ts @@ -25,7 +25,7 @@ describe('assembleScorecard', () => { (input) => assembleScorecard(input).rows.every((row) => (['ratstack', 'starter'] as const) - .filter((side) => !input.cells.some((cell) => cell.id === row.definition.id && cell.side === side)) + .filter((side) => input.cells[row.definition.id]?.[side] === undefined) .every((side) => row[side].cell._tag === 'InstrumentError' && row[side].cell.error.includes(row.definition.id) ) diff --git a/evals/ratstack-scorecard/src/model/scorecard-document.ts b/evals/ratstack-scorecard/src/model/scorecard-document.ts index e7e8ee1..9c49d18 100644 --- a/evals/ratstack-scorecard/src/model/scorecard-document.ts +++ b/evals/ratstack-scorecard/src/model/scorecard-document.ts @@ -27,9 +27,11 @@ export interface RowFlag { readonly flag: Flag } +export type CellsByRow = Readonly>>>> + export interface AssembleInput { readonly rows: readonly DefinedRow[] - readonly cells: readonly SideCell[] + readonly cells: CellsByRow readonly flags: readonly RowFlag[] readonly provenance: DocumentProvenance readonly main: MainBaseline @@ -49,10 +51,7 @@ const missingCell = (provenance: DocumentProvenance, id: string, side: Side): Me }) const cellFor = (input: AssembleInput, id: string, side: Side): MeasuredCell => - input.cells - .filter((cell) => cell.id === id) - .filter((cell) => cell.side === side) - .reduce((_, cell) => cell.measured, missingCell(input.provenance, id, side)) + input.cells[id]?.[side] ?? missingCell(input.provenance, id, side) const rowOf = (input: AssembleInput, defined: DefinedRow): Row => { const ratstack = cellFor(input, defined.definition.id, 'ratstack') diff --git a/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts b/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts index 3655133..411a908 100644 --- a/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts +++ b/evals/ratstack-scorecard/src/model/scorecard.arbitrary.ts @@ -34,6 +34,10 @@ export const definition = (overrides: Partial = {}): fc.Arbitrary 'agent-surfaces' as const, 'networked' as const, ), + ratstackSupport: fc.oneof( + fc.constant({ _tag: 'Required' } as const), + value.map((bar) => ({ _tag: 'MayBeUnsupported', bar }) as const), + ), }).map((generated) => ({ ...generated, ...overrides })) export const runsFor = (row: RowDefinition): fc.Arbitrary => @@ -121,7 +125,16 @@ export const assembleInput: fc.Arbitrary = uniqueDefinitions.chai fc.option(measuredCell(d, side).map((measured) => ({ id: d.id, side, measured })), { nil: undefined }) ) ), - ).map((cells) => cells.filter((c) => c !== undefined)), + ).map((cells) => + Object.fromEntries( + definitions.map((d) => [ + d.id, + Object.fromEntries( + cells.filter((c) => c !== undefined && c.id === d.id).map((c) => [c!.side, c!.measured]), + ), + ]), + ) + ), flags: fc.tuple( ...definitions.map((d) => fc.array(flag, { maxLength: 2 }).map((fs) => fs.map((f) => ({ id: d.id, flag: f })))), ) From b48af9be86898e38e1c3822ced2645c296a8ce70 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Tue, 6 Oct 2026 17:25:40 +0000 Subject: [PATCH 6/7] docs(repo): cite the superseding lake 1 plan from the scorecard plan starter-brainstorm superseded the Lake 1 plan with 2026-10-06-1703; the scorecard plan's source list now points at it, with the unit and ruling anchors it uses there --- docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md b/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md index 9aa43a0..8743ca2 100644 --- a/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md +++ b/docs/plans/2026-10-05-2151-feat-ratstack-scorecard-plan.md @@ -476,7 +476,7 @@ Mutation testing never runs locally; it runs only in the scorecard workflow's `m ## Sources - Origin: `docs/brainstorms/inputs/requirements-final.md` (Superiority Map; R46, R51, R68, R69), `docs/brainstorms/inputs/ruling-nix-distribution-sandbox.md`, `docs/brainstorms/inputs/2026-10-05-starter-ratstack-brief.md`. -- Lake 1 plan (sibling worktree `brainstorm`): `docs/plans/2026-10-05-2014-feat-starter-lake-1-foundation-plan.md`, KTD2 (fleet labels), KTD14 (local stack), U11b (sandbox), U13 (previews), Kiro Rulings item 6 (bubblewrap probe). +- Lake 1 plan (sibling worktree `brainstorm`): `docs/plans/2026-10-06-1703-feat-starter-lake-1-foundation-plan.md` (supersedes `2026-10-05-2014`), KTD2 (fleet labels), KTD14 (local stack), U11b (sandbox and bubblewrap probe), U13 (previews), Kiro Rulings item 6 (Nix distribution, consumers sandboxed). - rat-stack at `54d3560`: `package.json` (pnpm 11.3.0, `devEngines`), `.github/workflows/ci.yml:18-23`, `scripts/acceptance-cold-clone.sh`, `scripts/oxlint-plugin-debt-ledger.ts:6-13`, `apps/mischief/scripts/content-lib.ts:866-895`, `apps/mischief/src/app.ts:510-556`, `apps/mischief/src/worker.ts:96-166,363`, `apps/mischief/scripts/smoke.sh`, `packages/core/src/join-interest-contract.ts:22-26`, `apps/mischief/src/interest/interest-durable-object.ts:60-75`, README "Keep or cut". - Live: `https://ratstack.sh/debt.md` (217 directives), `https://ratstack.sh/log.md` (newest `ed63ba3`, 2026-10-03); `git ls-remote https://github.com/joelhooks/rat-stack` HEAD `54d356037c994f89698760a4727be71d0005a087`. - Versions verified 2026-10-05: npm `lighthouse` 13.5.0, `oxc-parser` 0.153.0, `yaml` 2.9.1, `fast-check` 4.10.2, `skills` 1.7.0, `pnpm` 11.3.0; nixpkgs `4975466d` `chromium` 153.0.8010.52, `claude-code` 2.1.280, `mitmproxy` 12.2.3, `nodejs_24` 24.20.0, `deno` 2.9.6, `bubblewrap` 0.12.0; GitHub `actions/upload-artifact` v7.0.1, `actions/checkout` v7.0.1. From cfb977b1a22ec363e30ee4d640b096d8f9e42ad0 Mon Sep 17 00:00:00 2001 From: Ryan Lee Date: Tue, 6 Oct 2026 17:50:39 +0000 Subject: [PATCH 7/7] test(repo): delete the direction mirror law that no longer fits the absolute bar Since the absolute bar for unsupported rat-stack rows, judging is no longer symmetric under negating runs and flipping direction: the bar does not negate with them, so the law failed on generated MaybeUnsupported definitions (about one run in four). The direction cases stay covered by the beaten, tie and instrument-error laws beside it; the same law is already gone from the CI layer --- .../src/model/judge-row.workflow.property.test.ts | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts b/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts index cbe7754..2f9472a 100644 --- a/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts +++ b/evals/ratstack-scorecard/src/model/judge-row.workflow.property.test.ts @@ -6,8 +6,6 @@ import { cell, definition, runsFor } from './scorecard.arbitrary.ts' const measured = (runs: readonly number[]): Cell => ({ _tag: 'Measured', runs }) -const negated = (c: Cell): Cell => c._tag === 'Measured' ? measured(c.runs.map((v) => -v)) : c - const verifiedCitation: Cell = { _tag: 'Unsupported', citation: { @@ -79,15 +77,6 @@ describe('judgeRow', () => { judgeRow({ definition: d, ratstack: measured(runs), starter: measured([...runs].reverse()) })._tag === 'Tie', ) - test.prop([definition().chain((d) => fc.tuple(fc.constant(d), cell(d), cell(d)))])( - 'direction higher mirrors direction lower under negation', - ([d, ratstack, starter]) => { - const flipped = { ...d, direction: ({ lower: 'higher', higher: 'lower' } as const)[d.direction] } - return judgeRow({ definition: flipped, ratstack: negated(ratstack), starter: negated(starter) })._tag === - judgeRow({ definition: d, ratstack, starter })._tag - }, - ) - test.prop([withRuns(), fc.integer({ min: 1, max: 4 }), fc.boolean()])( 'a side with fewer runs than its definition is an instrument error', ([d, runs], cut, cutStarter) => {