From e0ba162e087e1c877f9612876e1a18339579b117 Mon Sep 17 00:00:00 2001 From: Dmitrii Vasilev Date: Wed, 23 Sep 2026 19:13:18 +0700 Subject: [PATCH 1/3] feat(queen): WARS arena tab, compiled from the wars.t27 contract The queen screen gains a pipeline arena with four lanes: Bee baseline, Bee+JEV, IGLA CODER, IGLA RACE. wars.t27 is the single source of truth; wars.json and queenWars.generated.ts are verified projections, and unknown results render as em dashes, never zeros. JEV's lane states CREDENTIAL_BLOCKED until a real key exists - an audit result, not a loss. QA: queen-wars-contract.mjs forbids unknown executor models and unsealed logs; website-checks.yml runs it in CI. The grid is container-responsive after the 804px arena audit. --- .github/workflows/website-checks.yml | 15 + apps/website/package.json | 6 +- apps/website/public/queen/wars.json | 246 +++++++++ apps/website/public/queen/wars.t27 | 135 +++++ apps/website/qa/agents-spec-contract.mjs | 11 +- apps/website/qa/queen-contrast-contract.mjs | 2 + apps/website/qa/queen-viewport-contract.mjs | 8 +- apps/website/qa/queen-wars-contract.mjs | 109 ++++ apps/website/scripts/queen-wars-from-spec.mjs | 344 ++++++++++++ .../scripts/queen-wars-from-spec.test.mjs | 386 +++++++++++++ apps/website/specs/queen/wars.t27 | 135 +++++ apps/website/src/components/QueenWars.css | 513 ++++++++++++++++++ apps/website/src/components/QueenWars.tsx | 368 +++++++++++++ apps/website/src/components/queenHud.ts | 10 +- apps/website/src/lib/queenModules.ts | 17 + apps/website/src/lib/queenWars.generated.ts | 252 +++++++++ apps/website/src/pages/Queen.tsx | 8 + 17 files changed, 2554 insertions(+), 11 deletions(-) create mode 100644 apps/website/public/queen/wars.json create mode 100644 apps/website/public/queen/wars.t27 create mode 100644 apps/website/qa/queen-wars-contract.mjs create mode 100644 apps/website/scripts/queen-wars-from-spec.mjs create mode 100644 apps/website/scripts/queen-wars-from-spec.test.mjs create mode 100644 apps/website/specs/queen/wars.t27 create mode 100644 apps/website/src/components/QueenWars.css create mode 100644 apps/website/src/components/QueenWars.tsx create mode 100644 apps/website/src/lib/queenWars.generated.ts diff --git a/.github/workflows/website-checks.yml b/.github/workflows/website-checks.yml index b5331fdb59..92644517ad 100644 --- a/.github/workflows/website-checks.yml +++ b/.github/workflows/website-checks.yml @@ -256,6 +256,13 @@ jobs: - name: t27 evolution tree run: npm run check:t27-evolution + # WARS has exactly one source of truth: specs/queen/wars.t27. These gates + # execute its own assertions, reject unsupported or contradictory ledger + # rows, and prove that the public .t27, JSON and TypeScript projections + # are byte-for-byte current before Vite packages any of them. + - name: Queen WARS experiment ledger + run: npm run check:wars && npm run test:wars-spec + - name: Build run: npx vite build @@ -264,6 +271,14 @@ jobs: - uses: browser-actions/setup-chrome@v1 id: chrome + # Every Queen view, including WARS, must own its scrolling at desktop and + # phone widths and must remain stable with reduced motion. The build above + # is the exact artifact this browser contract inspects. + - name: Queen viewport matrix + env: + CHROME_PATH: ${{ steps.chrome.outputs.chrome-path }} + run: npm run check:queen-viewport -- --no-build + # Проверяет все маршруты в обеих локалях. Скрипты имеют собственные # самотесты и используют Chrome через CDP без новой зависимости npm. - name: Language audits diff --git a/apps/website/package.json b/apps/website/package.json index a933ac0668..0ac90ae092 100644 --- a/apps/website/package.json +++ b/apps/website/package.json @@ -16,8 +16,8 @@ "check:spec-catalog": "node --experimental-strip-types qa/spec-catalog-contract.mjs", "core": "node scripts/spec-core.mjs", "check:shared-core": "node --experimental-strip-types qa/shared-spec-core.mjs", - "prebuild": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs", - "prebuild:ci": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs", + "prebuild": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs && node scripts/queen-wars-from-spec.mjs", + "prebuild:ci": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs && node scripts/queen-wars-from-spec.mjs", "dev": "vite", "build": "vite build", "build:ci": "vite build", @@ -94,6 +94,8 @@ "test:docs-specs": "node --test scripts/docs-from-specs.test.mjs", "check:viewport": "node scripts/viewport-from-spec.mjs --check", "check:onboarding": "node scripts/onboarding-from-spec.mjs --check", + "check:wars": "node scripts/queen-wars-from-spec.mjs --check && node --experimental-strip-types qa/queen-wars-contract.mjs", + "test:wars-spec": "node --test scripts/queen-wars-from-spec.test.mjs", "test:viewport-spec": "node --test scripts/viewport-from-spec.test.mjs", "check:explorer-viewport": "node qa/explorer-viewport-contract.mjs", "check:passport": "node --experimental-strip-types qa/passport-figures.mjs" diff --git a/apps/website/public/queen/wars.json b/apps/website/public/queen/wars.json new file mode 100644 index 0000000000..71ec97fde5 --- /dev/null +++ b/apps/website/public/queen/wars.json @@ -0,0 +1,246 @@ +{ + "source": { + "spec": "specs/queen/wars.t27", + "publicSpec": "public/queen/wars.t27", + "sha256": "b7c1f22ea8361a092c083d0a2681477af2d7ec08bd35802dfd25289f0e784924", + "schemaVersion": 2 + }, + "name": "Queen WARS real-task agent arena", + "vocabularies": { + "evidence": [ + "OBSERVED", + "SESSION-OBSERVED", + "SOURCE-CLAIM", + "TARGET", + "UNKNOWN" + ], + "configStates": [ + "ready", + "credential-blocked", + "checkpoint-unverified", + "pipeline-only" + ], + "experimentStates": [ + "planned", + "running", + "credential-blocked", + "complete", + "invalid" + ], + "runStates": [ + "pending", + "running", + "passed", + "failed", + "blocked" + ], + "verdicts": [ + "accepted", + "rejected", + "inconclusive", + "not-reviewed" + ] + }, + "protocol": { + "realGitHubTasksOnly": true, + "variableFactor": "jev-decision-layer", + "controlledFactors": [ + "issue-url", + "base-sha", + "executor-model", + "prompt", + "tool-policy", + "budget", + "acceptance" + ], + "isolation": "one clean git worktree per arm, both pinned to the same base commit", + "acceptancePolicy": "the issue acceptance commands plus repository gates decide pass or fail", + "reviewPolicy": "Queen reviews the diff and exact evidence after automated gates", + "winnerPolicy": "merge at most one accepted patch; preserve every arm in the append-only ledger", + "comparisonValidityPolicy": "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner", + "jevRole": "structured decision layer that ranks typed choices; never a chat model and never a code generator", + "jevRoleEvidence": "SOURCE-CLAIM", + "jevRoleSource": "https://docs.typesafe.ai/introduction/coding-agents", + "emptyMetricMeansUnknown": true + }, + "metricCatalog": [ + { + "key": "acceptance", + "unit": "verdict" + }, + { + "key": "queen-verdict", + "unit": "verdict" + }, + { + "key": "elapsed", + "unit": "ms" + }, + { + "key": "input-tokens", + "unit": "tokens" + }, + { + "key": "output-tokens", + "unit": "tokens" + }, + { + "key": "tool-calls", + "unit": "count" + }, + { + "key": "retries", + "unit": "count" + }, + { + "key": "patch-lines", + "unit": "lines" + }, + { + "key": "cost", + "unit": "usd" + }, + { + "key": "jev-latency", + "unit": "ms" + }, + { + "key": "jev-confidence", + "unit": "probability" + } + ], + "configurations": [ + { + "id": "bee-baseline", + "name": "Bee baseline", + "kind": "coding-agent", + "state": "ready", + "evidence": "OBSERVED", + "source": "current Codex task runtime", + "note": "Control arm: the coding Bee chooses and implements without an external decision model.", + "stateEvidence": "OBSERVED", + "stateSource": "current Codex task runtime on 2026-09-23", + "stateNote": "The baseline Bee can run locally." + }, + { + "id": "bee-jev", + "name": "Bee + JEV", + "kind": "coding-agent-plus-decision-layer", + "state": "credential-blocked", + "evidence": "SOURCE-CLAIM", + "source": "https://docs.typesafe.ai/introduction/coding-agents", + "note": "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", + "stateEvidence": "OBSERVED", + "stateSource": "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", + "stateNote": "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed." + }, + { + "id": "igla-coder", + "name": "IGLA CODER", + "kind": "model-training-target", + "state": "checkpoint-unverified", + "evidence": "OBSERVED", + "source": "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", + "note": "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", + "stateEvidence": "OBSERVED", + "stateSource": "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", + "stateNote": "No executable checkpoint was verified for this arena." + }, + { + "id": "igla-race", + "name": "IGLA RACE", + "kind": "training-race-pipeline", + "state": "pipeline-only", + "evidence": "OBSERVED", + "source": "https://github.com/gHashTag/trios-trainer-igla", + "note": "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result.", + "stateEvidence": "OBSERVED", + "stateSource": "repository inspection of the linked IGLA RACE pipeline on 2026-09-23", + "stateNote": "The training pipeline is visible, but no coding-model run is claimed." + } + ], + "experiments": [ + { + "id": "t27-4328-validate-trits", + "name": "validate_trits missing test", + "issue": { + "repo": "gHashTag/t27", + "number": 4328, + "url": "https://github.com/gHashTag/t27/issues/4328", + "updatedAt": "2026-09-20T13:51:01Z" + }, + "baseSha": "f123674fe40d6ba600a8c5c4948683198f756fdc", + "executorModel": "UNKNOWN: exact model id and reasoning configuration were not exposed", + "prompt": "Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push.", + "promptSha256": "802e41621a28102a31ea6c64599c6af5723006cab168b65d196274e254effa9c", + "toolPolicy": "Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push.", + "toolPolicySha256": "1e8c7b9f7452453ec82eb0f3284f4d0881add2c7a6b99b07d72097b29d479246", + "budget": "one Codex agent turn; no explicit token cap; record usage only when exposed", + "acceptance": "t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass", + "acceptanceSha256": "7873d19d6dd591664a744423ba8ea2a9422e664b0e0558a633375b4c3edb3712", + "modelEvidence": "UNKNOWN", + "modelSource": "current Codex task runtime did not expose a stable model id and reasoning configuration", + "state": "credential-blocked", + "evidence": "OBSERVED", + "note": "This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed." + } + ], + "runs": [ + { + "id": "t27-4328-bee-baseline-20260923T041025Z", + "experimentId": "t27-4328-validate-trits", + "configId": "bee-baseline", + "startedAt": "2026-09-23T04:10:25Z", + "finishedAt": "2026-09-23T04:32:27Z", + "state": "blocked", + "evidence": "SESSION-OBSERVED", + "verdict": "inconclusive", + "logSha256": null, + "patchSha256": "11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267", + "artifactUrl": null, + "note": "The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed." + } + ], + "measurements": [ + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "acceptance", + "value": "BLOCKED", + "unit": "verdict", + "evidence": "SESSION-OBSERVED", + "source": "session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "queen-verdict", + "value": "inconclusive", + "unit": "verdict", + "evidence": "SESSION-OBSERVED", + "source": "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "elapsed", + "value": "1322000", + "unit": "ms", + "evidence": "SESSION-OBSERVED", + "source": "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "retries", + "value": "2", + "unit": "count", + "evidence": "SESSION-OBSERVED", + "source": "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "patch-lines", + "value": "9", + "unit": "lines", + "evidence": "SESSION-OBSERVED", + "source": "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS" + } + ] +} diff --git a/apps/website/public/queen/wars.t27 b/apps/website/public/queen/wars.t27 new file mode 100644 index 0000000000..503ea06e2a --- /dev/null +++ b/apps/website/public/queen/wars.t27 @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: Apache-2.0 +// specs/queen/wars.t27 -- source of truth for the Queen WARS experiment arena +// +// This module owns the experiment protocol, competitor configurations, real tasks, +// run ledger and measurements shown by the WARS view. React, JSON and the public +// copy of this file are generated projections. Missing measurements are absent; +// zero is never used to mean unknown. ASCII only (L3), English only (LANG-EN). +// phi^2 + 1/phi^2 = 3 | TRINITY + +module queen_wars; + +pub const KIND : str = "queen-wars"; +pub const ID : str = "queen/wars"; +pub const NAME : str = "Queen WARS real-task agent arena"; +pub const SCHEMA_VERSION : u8 = 2; +pub const GENERATED : [3]str = ["src/lib/queenWars.generated.ts", "public/queen/wars.json", "public/queen/wars.t27"]; + +// Epistemic and lifecycle vocabularies. Every displayed claim uses one of these. +pub const EVIDENCE_LEVELS : [5]str = ["OBSERVED", "SESSION-OBSERVED", "SOURCE-CLAIM", "TARGET", "UNKNOWN"]; +pub const CONFIG_STATES : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; +pub const EXPERIMENT_STATES : [5]str = ["planned", "running", "credential-blocked", "complete", "invalid"]; +pub const RUN_STATES : [5]str = ["pending", "running", "passed", "failed", "blocked"]; +pub const VERDICTS : [4]str = ["accepted", "rejected", "inconclusive", "not-reviewed"]; +pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; + +// Fairness contract. A result that breaks one controlled factor is not an A/B result. +pub const REAL_GITHUB_TASKS_ONLY : bool = true; +pub const VARIABLE_FACTOR : str = "jev-decision-layer"; +pub const CONTROLLED_FACTORS : [7]str = ["issue-url", "base-sha", "executor-model", "prompt", "tool-policy", "budget", "acceptance"]; +pub const ISOLATION : str = "one clean git worktree per arm, both pinned to the same base commit"; +pub const ACCEPTANCE_POLICY : str = "the issue acceptance commands plus repository gates decide pass or fail"; +pub const REVIEW_POLICY : str = "Queen reviews the diff and exact evidence after automated gates"; +pub const WINNER_POLICY : str = "merge at most one accepted patch; preserve every arm in the append-only ledger"; +pub const COMPARISON_VALIDITY_POLICY : str = "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner"; +pub const JEV_ROLE : str = "structured decision layer that ranks typed choices; never a chat model and never a code generator"; +pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM"; +pub const JEV_ROLE_SOURCE : str = "https://docs.typesafe.ai/introduction/coding-agents"; + +// Measurement vocabulary. Values are strings so an exact native unit is preserved; +// absent rows mean unknown. A numeric zero is a measured zero only when a source exists. +pub const METRIC_COUNT : u8 = 11; +pub const METRIC_KEYS : [11]str = ["acceptance", "queen-verdict", "elapsed", "input-tokens", "output-tokens", "tool-calls", "retries", "patch-lines", "cost", "jev-latency", "jev-confidence"]; +pub const METRIC_UNITS : [11]str = ["verdict", "verdict", "ms", "tokens", "tokens", "count", "count", "lines", "usd", "ms", "probability"]; + +// Arena configurations. IGLA entries are visible now, but neither is presented as a +// measured coding model until an executable checkpoint and a witnessed run exist. +pub const CONFIG_COUNT : u8 = 4; +pub const CONFIG_IDS : [4]str = ["bee-baseline", "bee-jev", "igla-coder", "igla-race"]; +pub const CONFIG_NAMES : [4]str = ["Bee baseline", "Bee + JEV", "IGLA CODER", "IGLA RACE"]; +pub const CONFIG_KINDS : [4]str = ["coding-agent", "coding-agent-plus-decision-layer", "model-training-target", "training-race-pipeline"]; +pub const CONFIG_STATES_BY_ID : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; +pub const CONFIG_EVIDENCE : [4]str = ["OBSERVED", "SOURCE-CLAIM", "OBSERVED", "OBSERVED"]; +pub const CONFIG_SOURCES : [4]str = ["current Codex task runtime", "https://docs.typesafe.ai/introduction/coding-agents", "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "https://github.com/gHashTag/trios-trainer-igla"]; +pub const CONFIG_NOTES : [4]str = ["Control arm: the coding Bee chooses and implements without an external decision model.", "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result."]; +pub const CONFIG_STATE_EVIDENCE : [4]str = ["OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const CONFIG_STATE_SOURCES : [4]str = ["current Codex task runtime on 2026-09-23", "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", "repository inspection of the linked IGLA RACE pipeline on 2026-09-23"]; +pub const CONFIG_STATE_NOTES : [4]str = ["The baseline Bee can run locally.", "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed.", "No executable checkpoint was verified for this arena.", "The training pipeline is visible, but no coding-model run is claimed."]; + +// Experiments. The exact prompt, tools, budget and acceptance live here, not in React. +pub const EXPERIMENT_COUNT : u8 = 1; +pub const EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; +pub const EXPERIMENT_NAMES : [1]str = ["validate_trits missing test"]; +pub const EXPERIMENT_REPOS : [1]str = ["gHashTag/t27"]; +pub const EXPERIMENT_ISSUE_URLS : [1]str = ["https://github.com/gHashTag/t27/issues/4328"]; +pub const EXPERIMENT_ISSUE_NUMBERS : [1]str = ["4328"]; +pub const EXPERIMENT_ISSUE_UPDATED_AT : [1]str = ["2026-09-20T13:51:01Z"]; +pub const EXPERIMENT_BASE_SHAS : [1]str = ["f123674fe40d6ba600a8c5c4948683198f756fdc"]; +pub const EXPERIMENT_EXECUTOR_MODELS : [1]str = ["UNKNOWN: exact model id and reasoning configuration were not exposed"]; +pub const EXPERIMENT_MODEL_EVIDENCE : [1]str = ["UNKNOWN"]; +pub const EXPERIMENT_MODEL_SOURCES : [1]str = ["current Codex task runtime did not expose a stable model id and reasoning configuration"]; +pub const EXPERIMENT_PROMPTS : [1]str = ["Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push."]; +pub const EXPERIMENT_TOOL_POLICIES : [1]str = ["Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push."]; +pub const EXPERIMENT_BUDGETS : [1]str = ["one Codex agent turn; no explicit token cap; record usage only when exposed"]; +pub const EXPERIMENT_ACCEPTANCE : [1]str = ["t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass"]; +pub const EXPERIMENT_STATES_BY_ID : [1]str = ["credential-blocked"]; +pub const EXPERIMENT_EVIDENCE : [1]str = ["OBSERVED"]; +pub const EXPERIMENT_NOTES : [1]str = ["This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed."]; + +// Append-only run ledger. Completed, failed and blocked attempts all stay visible. +pub const RUN_COUNT : u8 = 1; +pub const RUN_IDS : [1]str = ["t27-4328-bee-baseline-20260923T041025Z"]; +pub const RUN_EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; +pub const RUN_CONFIG_IDS : [1]str = ["bee-baseline"]; +pub const RUN_STARTED_AT : [1]str = ["2026-09-23T04:10:25Z"]; +pub const RUN_FINISHED_AT : [1]str = ["2026-09-23T04:32:27Z"]; +pub const RUN_STATES_BY_ID : [1]str = ["blocked"]; +pub const RUN_EVIDENCE : [1]str = ["SESSION-OBSERVED"]; +pub const RUN_VERDICTS : [1]str = ["inconclusive"]; +pub const RUN_LOG_SHAS : [1]str = [""]; +pub const RUN_PATCH_SHAS : [1]str = ["11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267"]; +pub const RUN_ARTIFACT_URLS : [1]str = [""]; +pub const RUN_NOTES : [1]str = ["The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed."]; + +// EAV measurements keep unknowns absent instead of turning them into misleading zeroes. +pub const MEASUREMENT_COUNT : u8 = 5; +pub const MEASUREMENT_RUN_IDS : [5]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z"]; +pub const MEASUREMENT_KEYS : [5]str = ["acceptance", "queen-verdict", "elapsed", "retries", "patch-lines"]; +pub const MEASUREMENT_VALUES : [5]str = ["BLOCKED", "inconclusive", "1322000", "2", "9"]; +pub const MEASUREMENT_UNITS : [5]str = ["verdict", "verdict", "ms", "count", "lines"]; +pub const MEASUREMENT_EVIDENCE : [5]str = ["SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED"]; +pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed", "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed", "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z", "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed", "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS"]; + +test protocol_has_one_variable { + assert REAL_GITHUB_TASKS_ONLY == true; + assert VARIABLE_FACTOR == "jev-decision-layer"; + assert CONTROLLED_FACTORS[0] == "issue-url"; + assert CONTROLLED_FACTORS[6] == "acceptance"; + assert EMPTY_METRIC_MEANS_UNKNOWN == true; + assert JEV_ROLE_EVIDENCE == "SOURCE-CLAIM"; +} + +test arena_names_all_four_lanes { + assert CONFIG_COUNT == 4; + assert CONFIG_IDS[0] == "bee-baseline"; + assert CONFIG_IDS[1] == "bee-jev"; + assert CONFIG_IDS[2] == "igla-coder"; + assert CONFIG_IDS[3] == "igla-race"; + assert CONFIG_EVIDENCE[1] == "SOURCE-CLAIM"; + assert CONFIG_STATE_EVIDENCE[1] == "OBSERVED"; +} + +test first_task_is_real_and_pinned { + assert EXPERIMENT_COUNT == 1; + assert EXPERIMENT_ISSUE_NUMBERS[0] == "4328"; + assert EXPERIMENT_REPOS[0] == "gHashTag/t27"; + assert EXPERIMENT_STATES_BY_ID[0] == "credential-blocked"; + assert EXPERIMENT_MODEL_EVIDENCE[0] == "UNKNOWN"; +} + +test metric_vocabulary_has_jev_and_cost_evidence { + assert METRIC_COUNT == 11; + assert METRIC_KEYS[8] == "cost"; + assert METRIC_KEYS[9] == "jev-latency"; + assert METRIC_KEYS[10] == "jev-confidence"; +} diff --git a/apps/website/qa/agents-spec-contract.mjs b/apps/website/qa/agents-spec-contract.mjs index 4110763543..a9c9f66e01 100644 --- a/apps/website/qa/agents-spec-contract.mjs +++ b/apps/website/qa/agents-spec-contract.mjs @@ -504,7 +504,12 @@ assert.ok(passportModule && HUD_VIEWS.includes('passport'), 'PASSPORT (the recor assert.equal(passportModule.key, 'b', 'PASSPORT opens on b: digits spent, t is TOOLS, p is PROJECT, r is TRI') assert.ok(passportModule.en.hint.includes('(key b)') && passportModule.ru.hint.includes('(клавиша b)'), 'PASSPORT names its letter key in both hints') for (const lang of ['en', 'ru']) assert.ok(passportModule[lang].name && passportModule[lang].body.length > 40, `passport: ${lang} copy missing`) -assert.equal(HUD_KEYS.slice(0, HUD_VIEWS.length).join(''), '1234567890tprbwm', 'the rail keys are 1-9, 0, t, p, r, b, w, m in that order') +const warsModule = MODULES.find((m) => m.tab === 'wars') +assert.ok(warsModule && HUD_VIEWS.includes('wars'), 'WARS (the real-task agent arena) is a module and a view') +assert.equal(warsModule.key, 'x', 'WARS opens on x: the crossed-blades key') +assert.ok(warsModule.en.hint.includes('(key x)') && warsModule.ru.hint.includes('(клавиша x)'), 'WARS names its letter key in both hints') +for (const lang of ['en', 'ru']) assert.ok(warsModule[lang].name && warsModule[lang].body.length > 40, `wars: ${lang} copy missing`) +assert.equal(HUD_KEYS.slice(0, HUD_VIEWS.length).join(''), '1234567890tprbwmx', 'the rail keys are 1-9, 0, t, p, r, b, w, m, x in that order') // The rail is no longer the whole vocabulary. HUD_VIEWS stays the fourteen // addresses -- every ?tab=, every key, every module card -- while the rail draws @@ -525,8 +530,8 @@ assert.deepEqual( ) assert.deepEqual( [...BOARD_VIEWS], - ['kanban', 'map', 'factory'], - 'the board is kanban, mission map, factory -- the order of their keys, 3/4/5', + ['kanban', 'map', 'factory', 'research'], + 'the board is kanban, mission map, factory, tech tree -- the order of their keys, 3/4/5/6', ) assert.deepEqual( [...new Set([...RAIL_VIEWS, ...SPEC_LAYERS, ...BOARD_VIEWS])].sort(), diff --git a/apps/website/qa/queen-contrast-contract.mjs b/apps/website/qa/queen-contrast-contract.mjs index a8c83044a2..b1762840c8 100644 --- a/apps/website/qa/queen-contrast-contract.mjs +++ b/apps/website/qa/queen-contrast-contract.mjs @@ -200,6 +200,7 @@ const REACHED = { 'src/components/QueenSpecTreasury.css', 'src/components/QueenTri.css', 'src/components/QueenUniverseAtlas.css', + 'src/components/QueenWars.css', 'src/pages/Queen.css', 'src/pages/QueenUniverse.css', 'src/pages/passport.css', @@ -826,6 +827,7 @@ const TOKEN_SCOPES = new Map([ ['.queen-catalog-layer', 'board'], ['.queen-catalog-layer:has(.queen-catalog-toolbar.is-search-open)', 'board'], ['.queen-hive-display', 'board'], + ['.queen-wars', 'board'], ['.queen27-context', 'board'], ['.queen27-cycle-brand', 'board'], ['.queen27-factory', 'board'], diff --git a/apps/website/qa/queen-viewport-contract.mjs b/apps/website/qa/queen-viewport-contract.mjs index 3b812c76af..85c021eccf 100644 --- a/apps/website/qa/queen-viewport-contract.mjs +++ b/apps/website/qa/queen-viewport-contract.mjs @@ -28,7 +28,7 @@ const DIST = join(ROOT, 'dist'); const ROUTE = '#/queen'; const SHOTS = '/tmp/hud-shots'; const SIZES = [[1920, 1080], [1440, 900], [1272, 806], [1280, 700], [1280, 600], [390, 844]]; -const VIEWS = ['comb', 'kanban', 'map', 'factory', 'research']; +const VIEWS = ['comb', 'kanban', 'map', 'factory', 'research', 'wars']; // The rail used to draw one button per view, so this counted HUD_VIEWS and // compared. That stopped being the shape of the thing: SPECS has long stood for // six layers behind one button, and KANBAN now stands for itself, MAP and @@ -51,10 +51,11 @@ const readList = (name, pattern) => { const HUD_VIEWS = readList('HUD_VIEWS', /export const HUD_VIEWS[^=]*=\s*\[([\s\S]*?)\]\s*as const/); const SPEC_LAYERS = readList('SPEC_LAYERS', /export const SPEC_LAYERS\s*=\s*\[([\s\S]*?)\]\s*as const/); const BOARD_VIEWS = readList('BOARD_VIEWS', /export const BOARD_VIEWS\s*=\s*\[([\s\S]*?)\]\s*as const/); +const PROJECT_VIEWS = readList('PROJECT_VIEWS', /export const PROJECT_VIEWS\s*=\s*\[([\s\S]*?)\]\s*as const/); // A family's first entry is the button; the rest are behind it. This mirrors // isFolded in queenHud.ts, which is the one place the app decides it. const FOLD = new Map(); -for (const family of [SPEC_LAYERS, BOARD_VIEWS]) { +for (const family of [SPEC_LAYERS, BOARD_VIEWS, PROJECT_VIEWS]) { for (const view of family.slice(1)) FOLD.set(view, family[0]); } const RAIL_VIEWS = HUD_VIEWS.filter((view) => !FOLD.has(view)); @@ -224,6 +225,7 @@ const DECLARED = [ '.queen27-city-build-queue ol', '.queen27-hardware-foundry ol', '.queen27-city-console ol', '.queen27-city-head dl', '.queen27-city-build-queue dl', '.queen27-hardware-foundry dl', '.queen27-factory-command dl', '.queen27-activity-stream ol', '.queen27-flow-grid', + '.queen-wars', '.queen-wars-table-scroll', '.queen-wars-flow', // the sub-navigation row: one line at every width, scrolling sideways when // the rungs are wider than the module -- overflow-x:auto with the bar hidden, // which is the declaration. This gate never met one before, because the row @@ -269,7 +271,7 @@ const PROBE = (phone) => `(() => { // arrive as a single child. Matching only direct children counted zero views // on any tab that had grown a row above it, and reported a rendered board as // a board that had not rendered at all. - const viewSel = '.queen27-comb, .queen27-kanban, .queen27-mission-map, .queen27-factory, .queen27-tech'; + const viewSel = '.queen27-comb, .queen27-kanban, .queen27-mission-map, .queen27-factory, .queen27-tech, .queen-wars'; const views = body ? [...body.children].flatMap(child => child.matches(viewSel) ? [child] : [...child.querySelectorAll(':scope > ' + viewSel)]) diff --git a/apps/website/qa/queen-wars-contract.mjs b/apps/website/qa/queen-wars-contract.mjs new file mode 100644 index 0000000000..399824201c --- /dev/null +++ b/apps/website/qa/queen-wars-contract.mjs @@ -0,0 +1,109 @@ +// Contract for the WARS arena. The experiment protocol and every result live in +// specs/queen/wars.t27; the UI and public files are projections, never another ledger. +import assert from 'node:assert/strict' +import { existsSync, readFileSync } from 'node:fs' +import { join } from 'node:path' + +const ROOT = new URL('..', import.meta.url).pathname +const read = (rel) => readFileSync(join(ROOT, rel), 'utf8') +const mustExist = (rel) => assert.ok(existsSync(join(ROOT, rel)), `${rel} is missing`) + +const SPEC = 'specs/queen/wars.t27' +const GENERATOR = 'scripts/queen-wars-from-spec.mjs' +const TS = 'src/lib/queenWars.generated.ts' +const JSON_OUT = 'public/queen/wars.json' +const PUBLIC_SPEC = 'public/queen/wars.t27' +const COMPONENT = 'src/components/QueenWars.tsx' +const CSS = 'src/components/QueenWars.css' + +for (const file of [SPEC, GENERATOR, TS, JSON_OUT, PUBLIC_SPEC, COMPONENT, CSS]) mustExist(file) + +const spec = read(SPEC) +assert.doesNotMatch(spec, /[^\x00-\x7f]/, 'the WARS .t27 source must remain ASCII/L3') +assert.match(spec, /pub const REAL_GITHUB_TASKS_ONLY : bool = true;/) +assert.match(spec, /pub const VARIABLE_FACTOR : str = "jev-decision-layer";/) +assert.match(spec, /pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM";/) +assert.match(spec, /https:\/\/docs\.typesafe\.ai\/introduction\/coding-agents/) +assert.match(spec, /https:\/\/github\.com\/gHashTag\/t27\/issues\/4328/) +assert.match(spec, /current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth/) +assert.match(spec, /production Railway variables were not inspected/) +assert.match(spec, /pub const RUN_COUNT : u8 = \d+;/) +assert.match(spec, /pub const MEASUREMENT_COUNT : u8 = \d+;/) + +const generated = JSON.parse(read(JSON_OUT)) +assert.equal(generated.source.spec, SPEC) +assert.match(generated.source.sha256, /^[0-9a-f]{64}$/) +assert.deepEqual(generated.configurations.map((item) => item.id), ['bee-baseline', 'bee-jev', 'igla-coder', 'igla-race']) +assert.equal(generated.protocol.variableFactor, 'jev-decision-layer') +assert.equal(generated.protocol.jevRoleEvidence, 'SOURCE-CLAIM') +assert.match(generated.protocol.jevRoleSource, /^https:\/\/docs\.typesafe\.ai\//) +assert.equal(generated.configurations[1].evidence, 'SOURCE-CLAIM') +assert.equal(generated.configurations[1].stateEvidence, 'OBSERVED') +assert.equal(generated.experiments[0].issue.url, 'https://github.com/gHashTag/t27/issues/4328') +assert.equal(generated.experiments[0].baseSha.length, 40) +assert.equal(generated.experiments[0].modelEvidence, 'UNKNOWN') +for (const measurement of generated.measurements) { + assert.ok(measurement.source, `${measurement.runId}/${measurement.key} has evidence but no source`) +} + +assert.equal(read(PUBLIC_SPEC), spec, 'public/queen/wars.t27 must be byte-identical to the source spec') + +const hud = read('src/components/queenHud.ts') +assert.match(hud, /"wars"/) +assert.match(hud, /"KeyX"/) +const shell = read('src/pages/Queen.tsx') +assert.match(shell, /import \{ QueenWars \}/) +assert.match(shell, /view: "wars" as const/) +assert.match(shell, /boardView === "wars"/) +assert.match(shell, /warsView: "WARS"/) +assert.match(shell, /warsView: "ВОЙНЫ"/) + +const component = read(COMPONENT) +assert.match(component, /QUEEN_WARS/) +assert.match(component, /const \[selectedExperimentId, setSelectedExperimentId\] = useState\(QUEEN_WARS\.experiments\[0\]\.id\)/) +assert.match(component, /QUEEN_WARS\.experiments\.find\(\(item\) => item\.id === selectedExperimentId\)/) +assert.match(component, /