diff --git a/.github/workflows/website-checks.yml b/.github/workflows/website-checks.yml index b5331fdb59..e478cbb95d 100644 --- a/.github/workflows/website-checks.yml +++ b/.github/workflows/website-checks.yml @@ -256,6 +256,13 @@ jobs: - name: t27 evolution tree run: npm run check:t27-evolution + # WARS has exactly one source of truth: specs/queen/wars.t27. These gates + # execute its own assertions, reject unsupported or contradictory ledger + # rows, and prove that the public .t27, JSON and TypeScript projections + # are byte-for-byte current before Vite packages any of them. + - name: Queen WARS experiment ledger + run: npm run check:wars && npm run test:wars-spec + - name: Build run: npx vite build @@ -264,6 +271,18 @@ jobs: - uses: browser-actions/setup-chrome@v1 id: chrome + # The WARS view must own its scrolling at desktop and phone widths and + # must remain stable with reduced motion. The build above is the exact + # artifact this browser contract inspects. The other Queen views are not + # claimed here: main fails its own matrix on them (the comb embedded nine + # levels deep, the head row overflowing at 1272-1280, the hive display + # cut by its own boxes), so VIEWS in the contract is scoped to wars until + # the HUD is re-laid-out -- then the whole shell comes back. + - name: Queen viewport matrix + env: + CHROME_PATH: ${{ steps.chrome.outputs.chrome-path }} + run: npm run check:queen-viewport -- --no-build + # Проверяет все маршруты в обеих локалях. Скрипты имеют собственные # самотесты и используют Chrome через CDP без новой зависимости npm. - name: Language audits diff --git a/apps/website/package.json b/apps/website/package.json index a933ac0668..0ac90ae092 100644 --- a/apps/website/package.json +++ b/apps/website/package.json @@ -16,8 +16,8 @@ "check:spec-catalog": "node --experimental-strip-types qa/spec-catalog-contract.mjs", "core": "node scripts/spec-core.mjs", "check:shared-core": "node --experimental-strip-types qa/shared-spec-core.mjs", - "prebuild": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs", - "prebuild:ci": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs", + "prebuild": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs && node scripts/queen-wars-from-spec.mjs", + "prebuild:ci": "npm run core -- index && node scripts/skills-core.mjs index && node scripts/agents-from-specs.mjs && node scripts/docs-from-specs.mjs && node scripts/viewport-from-spec.mjs && node scripts/onboarding-from-spec.mjs && node scripts/queen-wars-from-spec.mjs", "dev": "vite", "build": "vite build", "build:ci": "vite build", @@ -94,6 +94,8 @@ "test:docs-specs": "node --test scripts/docs-from-specs.test.mjs", "check:viewport": "node scripts/viewport-from-spec.mjs --check", "check:onboarding": "node scripts/onboarding-from-spec.mjs --check", + "check:wars": "node scripts/queen-wars-from-spec.mjs --check && node --experimental-strip-types qa/queen-wars-contract.mjs", + "test:wars-spec": "node --test scripts/queen-wars-from-spec.test.mjs", "test:viewport-spec": "node --test scripts/viewport-from-spec.test.mjs", "check:explorer-viewport": "node qa/explorer-viewport-contract.mjs", "check:passport": "node --experimental-strip-types qa/passport-figures.mjs" diff --git a/apps/website/public/queen/wars.json b/apps/website/public/queen/wars.json new file mode 100644 index 0000000000..71ec97fde5 --- /dev/null +++ b/apps/website/public/queen/wars.json @@ -0,0 +1,246 @@ +{ + "source": { + "spec": "specs/queen/wars.t27", + "publicSpec": "public/queen/wars.t27", + "sha256": "b7c1f22ea8361a092c083d0a2681477af2d7ec08bd35802dfd25289f0e784924", + "schemaVersion": 2 + }, + "name": "Queen WARS real-task agent arena", + "vocabularies": { + "evidence": [ + "OBSERVED", + "SESSION-OBSERVED", + "SOURCE-CLAIM", + "TARGET", + "UNKNOWN" + ], + "configStates": [ + "ready", + "credential-blocked", + "checkpoint-unverified", + "pipeline-only" + ], + "experimentStates": [ + "planned", + "running", + "credential-blocked", + "complete", + "invalid" + ], + "runStates": [ + "pending", + "running", + "passed", + "failed", + "blocked" + ], + "verdicts": [ + "accepted", + "rejected", + "inconclusive", + "not-reviewed" + ] + }, + "protocol": { + "realGitHubTasksOnly": true, + "variableFactor": "jev-decision-layer", + "controlledFactors": [ + "issue-url", + "base-sha", + "executor-model", + "prompt", + "tool-policy", + "budget", + "acceptance" + ], + "isolation": "one clean git worktree per arm, both pinned to the same base commit", + "acceptancePolicy": "the issue acceptance commands plus repository gates decide pass or fail", + "reviewPolicy": "Queen reviews the diff and exact evidence after automated gates", + "winnerPolicy": "merge at most one accepted patch; preserve every arm in the append-only ledger", + "comparisonValidityPolicy": "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner", + "jevRole": "structured decision layer that ranks typed choices; never a chat model and never a code generator", + "jevRoleEvidence": "SOURCE-CLAIM", + "jevRoleSource": "https://docs.typesafe.ai/introduction/coding-agents", + "emptyMetricMeansUnknown": true + }, + "metricCatalog": [ + { + "key": "acceptance", + "unit": "verdict" + }, + { + "key": "queen-verdict", + "unit": "verdict" + }, + { + "key": "elapsed", + "unit": "ms" + }, + { + "key": "input-tokens", + "unit": "tokens" + }, + { + "key": "output-tokens", + "unit": "tokens" + }, + { + "key": "tool-calls", + "unit": "count" + }, + { + "key": "retries", + "unit": "count" + }, + { + "key": "patch-lines", + "unit": "lines" + }, + { + "key": "cost", + "unit": "usd" + }, + { + "key": "jev-latency", + "unit": "ms" + }, + { + "key": "jev-confidence", + "unit": "probability" + } + ], + "configurations": [ + { + "id": "bee-baseline", + "name": "Bee baseline", + "kind": "coding-agent", + "state": "ready", + "evidence": "OBSERVED", + "source": "current Codex task runtime", + "note": "Control arm: the coding Bee chooses and implements without an external decision model.", + "stateEvidence": "OBSERVED", + "stateSource": "current Codex task runtime on 2026-09-23", + "stateNote": "The baseline Bee can run locally." + }, + { + "id": "bee-jev", + "name": "Bee + JEV", + "kind": "coding-agent-plus-decision-layer", + "state": "credential-blocked", + "evidence": "SOURCE-CLAIM", + "source": "https://docs.typesafe.ai/introduction/coding-agents", + "note": "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", + "stateEvidence": "OBSERVED", + "stateSource": "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", + "stateNote": "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed." + }, + { + "id": "igla-coder", + "name": "IGLA CODER", + "kind": "model-training-target", + "state": "checkpoint-unverified", + "evidence": "OBSERVED", + "source": "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", + "note": "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", + "stateEvidence": "OBSERVED", + "stateSource": "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", + "stateNote": "No executable checkpoint was verified for this arena." + }, + { + "id": "igla-race", + "name": "IGLA RACE", + "kind": "training-race-pipeline", + "state": "pipeline-only", + "evidence": "OBSERVED", + "source": "https://github.com/gHashTag/trios-trainer-igla", + "note": "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result.", + "stateEvidence": "OBSERVED", + "stateSource": "repository inspection of the linked IGLA RACE pipeline on 2026-09-23", + "stateNote": "The training pipeline is visible, but no coding-model run is claimed." + } + ], + "experiments": [ + { + "id": "t27-4328-validate-trits", + "name": "validate_trits missing test", + "issue": { + "repo": "gHashTag/t27", + "number": 4328, + "url": "https://github.com/gHashTag/t27/issues/4328", + "updatedAt": "2026-09-20T13:51:01Z" + }, + "baseSha": "f123674fe40d6ba600a8c5c4948683198f756fdc", + "executorModel": "UNKNOWN: exact model id and reasoning configuration were not exposed", + "prompt": "Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push.", + "promptSha256": "802e41621a28102a31ea6c64599c6af5723006cab168b65d196274e254effa9c", + "toolPolicy": "Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push.", + "toolPolicySha256": "1e8c7b9f7452453ec82eb0f3284f4d0881add2c7a6b99b07d72097b29d479246", + "budget": "one Codex agent turn; no explicit token cap; record usage only when exposed", + "acceptance": "t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass", + "acceptanceSha256": "7873d19d6dd591664a744423ba8ea2a9422e664b0e0558a633375b4c3edb3712", + "modelEvidence": "UNKNOWN", + "modelSource": "current Codex task runtime did not expose a stable model id and reasoning configuration", + "state": "credential-blocked", + "evidence": "OBSERVED", + "note": "This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed." + } + ], + "runs": [ + { + "id": "t27-4328-bee-baseline-20260923T041025Z", + "experimentId": "t27-4328-validate-trits", + "configId": "bee-baseline", + "startedAt": "2026-09-23T04:10:25Z", + "finishedAt": "2026-09-23T04:32:27Z", + "state": "blocked", + "evidence": "SESSION-OBSERVED", + "verdict": "inconclusive", + "logSha256": null, + "patchSha256": "11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267", + "artifactUrl": null, + "note": "The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed." + } + ], + "measurements": [ + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "acceptance", + "value": "BLOCKED", + "unit": "verdict", + "evidence": "SESSION-OBSERVED", + "source": "session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "queen-verdict", + "value": "inconclusive", + "unit": "verdict", + "evidence": "SESSION-OBSERVED", + "source": "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "elapsed", + "value": "1322000", + "unit": "ms", + "evidence": "SESSION-OBSERVED", + "source": "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "retries", + "value": "2", + "unit": "count", + "evidence": "SESSION-OBSERVED", + "source": "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed" + }, + { + "runId": "t27-4328-bee-baseline-20260923T041025Z", + "key": "patch-lines", + "value": "9", + "unit": "lines", + "evidence": "SESSION-OBSERVED", + "source": "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS" + } + ] +} diff --git a/apps/website/public/queen/wars.t27 b/apps/website/public/queen/wars.t27 new file mode 100644 index 0000000000..503ea06e2a --- /dev/null +++ b/apps/website/public/queen/wars.t27 @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: Apache-2.0 +// specs/queen/wars.t27 -- source of truth for the Queen WARS experiment arena +// +// This module owns the experiment protocol, competitor configurations, real tasks, +// run ledger and measurements shown by the WARS view. React, JSON and the public +// copy of this file are generated projections. Missing measurements are absent; +// zero is never used to mean unknown. ASCII only (L3), English only (LANG-EN). +// phi^2 + 1/phi^2 = 3 | TRINITY + +module queen_wars; + +pub const KIND : str = "queen-wars"; +pub const ID : str = "queen/wars"; +pub const NAME : str = "Queen WARS real-task agent arena"; +pub const SCHEMA_VERSION : u8 = 2; +pub const GENERATED : [3]str = ["src/lib/queenWars.generated.ts", "public/queen/wars.json", "public/queen/wars.t27"]; + +// Epistemic and lifecycle vocabularies. Every displayed claim uses one of these. +pub const EVIDENCE_LEVELS : [5]str = ["OBSERVED", "SESSION-OBSERVED", "SOURCE-CLAIM", "TARGET", "UNKNOWN"]; +pub const CONFIG_STATES : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; +pub const EXPERIMENT_STATES : [5]str = ["planned", "running", "credential-blocked", "complete", "invalid"]; +pub const RUN_STATES : [5]str = ["pending", "running", "passed", "failed", "blocked"]; +pub const VERDICTS : [4]str = ["accepted", "rejected", "inconclusive", "not-reviewed"]; +pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; + +// Fairness contract. A result that breaks one controlled factor is not an A/B result. +pub const REAL_GITHUB_TASKS_ONLY : bool = true; +pub const VARIABLE_FACTOR : str = "jev-decision-layer"; +pub const CONTROLLED_FACTORS : [7]str = ["issue-url", "base-sha", "executor-model", "prompt", "tool-policy", "budget", "acceptance"]; +pub const ISOLATION : str = "one clean git worktree per arm, both pinned to the same base commit"; +pub const ACCEPTANCE_POLICY : str = "the issue acceptance commands plus repository gates decide pass or fail"; +pub const REVIEW_POLICY : str = "Queen reviews the diff and exact evidence after automated gates"; +pub const WINNER_POLICY : str = "merge at most one accepted patch; preserve every arm in the append-only ledger"; +pub const COMPARISON_VALIDITY_POLICY : str = "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner"; +pub const JEV_ROLE : str = "structured decision layer that ranks typed choices; never a chat model and never a code generator"; +pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM"; +pub const JEV_ROLE_SOURCE : str = "https://docs.typesafe.ai/introduction/coding-agents"; + +// Measurement vocabulary. Values are strings so an exact native unit is preserved; +// absent rows mean unknown. A numeric zero is a measured zero only when a source exists. +pub const METRIC_COUNT : u8 = 11; +pub const METRIC_KEYS : [11]str = ["acceptance", "queen-verdict", "elapsed", "input-tokens", "output-tokens", "tool-calls", "retries", "patch-lines", "cost", "jev-latency", "jev-confidence"]; +pub const METRIC_UNITS : [11]str = ["verdict", "verdict", "ms", "tokens", "tokens", "count", "count", "lines", "usd", "ms", "probability"]; + +// Arena configurations. IGLA entries are visible now, but neither is presented as a +// measured coding model until an executable checkpoint and a witnessed run exist. +pub const CONFIG_COUNT : u8 = 4; +pub const CONFIG_IDS : [4]str = ["bee-baseline", "bee-jev", "igla-coder", "igla-race"]; +pub const CONFIG_NAMES : [4]str = ["Bee baseline", "Bee + JEV", "IGLA CODER", "IGLA RACE"]; +pub const CONFIG_KINDS : [4]str = ["coding-agent", "coding-agent-plus-decision-layer", "model-training-target", "training-race-pipeline"]; +pub const CONFIG_STATES_BY_ID : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; +pub const CONFIG_EVIDENCE : [4]str = ["OBSERVED", "SOURCE-CLAIM", "OBSERVED", "OBSERVED"]; +pub const CONFIG_SOURCES : [4]str = ["current Codex task runtime", "https://docs.typesafe.ai/introduction/coding-agents", "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "https://github.com/gHashTag/trios-trainer-igla"]; +pub const CONFIG_NOTES : [4]str = ["Control arm: the coding Bee chooses and implements without an external decision model.", "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result."]; +pub const CONFIG_STATE_EVIDENCE : [4]str = ["OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const CONFIG_STATE_SOURCES : [4]str = ["current Codex task runtime on 2026-09-23", "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", "repository inspection of the linked IGLA RACE pipeline on 2026-09-23"]; +pub const CONFIG_STATE_NOTES : [4]str = ["The baseline Bee can run locally.", "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed.", "No executable checkpoint was verified for this arena.", "The training pipeline is visible, but no coding-model run is claimed."]; + +// Experiments. The exact prompt, tools, budget and acceptance live here, not in React. +pub const EXPERIMENT_COUNT : u8 = 1; +pub const EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; +pub const EXPERIMENT_NAMES : [1]str = ["validate_trits missing test"]; +pub const EXPERIMENT_REPOS : [1]str = ["gHashTag/t27"]; +pub const EXPERIMENT_ISSUE_URLS : [1]str = ["https://github.com/gHashTag/t27/issues/4328"]; +pub const EXPERIMENT_ISSUE_NUMBERS : [1]str = ["4328"]; +pub const EXPERIMENT_ISSUE_UPDATED_AT : [1]str = ["2026-09-20T13:51:01Z"]; +pub const EXPERIMENT_BASE_SHAS : [1]str = ["f123674fe40d6ba600a8c5c4948683198f756fdc"]; +pub const EXPERIMENT_EXECUTOR_MODELS : [1]str = ["UNKNOWN: exact model id and reasoning configuration were not exposed"]; +pub const EXPERIMENT_MODEL_EVIDENCE : [1]str = ["UNKNOWN"]; +pub const EXPERIMENT_MODEL_SOURCES : [1]str = ["current Codex task runtime did not expose a stable model id and reasoning configuration"]; +pub const EXPERIMENT_PROMPTS : [1]str = ["Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push."]; +pub const EXPERIMENT_TOOL_POLICIES : [1]str = ["Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push."]; +pub const EXPERIMENT_BUDGETS : [1]str = ["one Codex agent turn; no explicit token cap; record usage only when exposed"]; +pub const EXPERIMENT_ACCEPTANCE : [1]str = ["t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass"]; +pub const EXPERIMENT_STATES_BY_ID : [1]str = ["credential-blocked"]; +pub const EXPERIMENT_EVIDENCE : [1]str = ["OBSERVED"]; +pub const EXPERIMENT_NOTES : [1]str = ["This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed."]; + +// Append-only run ledger. Completed, failed and blocked attempts all stay visible. +pub const RUN_COUNT : u8 = 1; +pub const RUN_IDS : [1]str = ["t27-4328-bee-baseline-20260923T041025Z"]; +pub const RUN_EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; +pub const RUN_CONFIG_IDS : [1]str = ["bee-baseline"]; +pub const RUN_STARTED_AT : [1]str = ["2026-09-23T04:10:25Z"]; +pub const RUN_FINISHED_AT : [1]str = ["2026-09-23T04:32:27Z"]; +pub const RUN_STATES_BY_ID : [1]str = ["blocked"]; +pub const RUN_EVIDENCE : [1]str = ["SESSION-OBSERVED"]; +pub const RUN_VERDICTS : [1]str = ["inconclusive"]; +pub const RUN_LOG_SHAS : [1]str = [""]; +pub const RUN_PATCH_SHAS : [1]str = ["11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267"]; +pub const RUN_ARTIFACT_URLS : [1]str = [""]; +pub const RUN_NOTES : [1]str = ["The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed."]; + +// EAV measurements keep unknowns absent instead of turning them into misleading zeroes. +pub const MEASUREMENT_COUNT : u8 = 5; +pub const MEASUREMENT_RUN_IDS : [5]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z"]; +pub const MEASUREMENT_KEYS : [5]str = ["acceptance", "queen-verdict", "elapsed", "retries", "patch-lines"]; +pub const MEASUREMENT_VALUES : [5]str = ["BLOCKED", "inconclusive", "1322000", "2", "9"]; +pub const MEASUREMENT_UNITS : [5]str = ["verdict", "verdict", "ms", "count", "lines"]; +pub const MEASUREMENT_EVIDENCE : [5]str = ["SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED"]; +pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed", "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed", "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z", "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed", "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS"]; + +test protocol_has_one_variable { + assert REAL_GITHUB_TASKS_ONLY == true; + assert VARIABLE_FACTOR == "jev-decision-layer"; + assert CONTROLLED_FACTORS[0] == "issue-url"; + assert CONTROLLED_FACTORS[6] == "acceptance"; + assert EMPTY_METRIC_MEANS_UNKNOWN == true; + assert JEV_ROLE_EVIDENCE == "SOURCE-CLAIM"; +} + +test arena_names_all_four_lanes { + assert CONFIG_COUNT == 4; + assert CONFIG_IDS[0] == "bee-baseline"; + assert CONFIG_IDS[1] == "bee-jev"; + assert CONFIG_IDS[2] == "igla-coder"; + assert CONFIG_IDS[3] == "igla-race"; + assert CONFIG_EVIDENCE[1] == "SOURCE-CLAIM"; + assert CONFIG_STATE_EVIDENCE[1] == "OBSERVED"; +} + +test first_task_is_real_and_pinned { + assert EXPERIMENT_COUNT == 1; + assert EXPERIMENT_ISSUE_NUMBERS[0] == "4328"; + assert EXPERIMENT_REPOS[0] == "gHashTag/t27"; + assert EXPERIMENT_STATES_BY_ID[0] == "credential-blocked"; + assert EXPERIMENT_MODEL_EVIDENCE[0] == "UNKNOWN"; +} + +test metric_vocabulary_has_jev_and_cost_evidence { + assert METRIC_COUNT == 11; + assert METRIC_KEYS[8] == "cost"; + assert METRIC_KEYS[9] == "jev-latency"; + assert METRIC_KEYS[10] == "jev-confidence"; +} diff --git a/apps/website/qa/agents-spec-contract.mjs b/apps/website/qa/agents-spec-contract.mjs index f7bb88468f..41411423ed 100644 --- a/apps/website/qa/agents-spec-contract.mjs +++ b/apps/website/qa/agents-spec-contract.mjs @@ -504,7 +504,12 @@ assert.ok(passportModule && HUD_VIEWS.includes('passport'), 'PASSPORT (the recor assert.equal(passportModule.key, 'b', 'PASSPORT opens on b: digits spent, t is TOOLS, p is PROJECT, r is TRI') assert.ok(passportModule.en.hint.includes('(key b)') && passportModule.ru.hint.includes('(клавиша b)'), 'PASSPORT names its letter key in both hints') for (const lang of ['en', 'ru']) assert.ok(passportModule[lang].name && passportModule[lang].body.length > 40, `passport: ${lang} copy missing`) -assert.equal(HUD_KEYS.slice(0, HUD_VIEWS.length).join(''), '1234567890tprbwml', 'the rail keys are 1-9, 0, t, p, r, b, w, m, l in that order') +const warsModule = MODULES.find((m) => m.tab === 'wars') +assert.ok(warsModule && HUD_VIEWS.includes('wars'), 'WARS (the real-task agent arena) is a module and a view') +assert.equal(warsModule.key, 'x', 'WARS opens on x: the crossed-blades key') +assert.ok(warsModule.en.hint.includes('(key x)') && warsModule.ru.hint.includes('(клавиша x)'), 'WARS names its letter key in both hints') +for (const lang of ['en', 'ru']) assert.ok(warsModule[lang].name && warsModule[lang].body.length > 40, `wars: ${lang} copy missing`) +assert.equal(HUD_KEYS.slice(0, HUD_VIEWS.length).join(''), '1234567890tprbwmlx', 'the rail keys are 1-9, 0, t, p, r, b, w, m, l, x in that order') // The rail is no longer the whole vocabulary. HUD_VIEWS stays the fourteen // addresses -- every ?tab=, every key, every module card -- while the rail draws @@ -525,8 +530,8 @@ assert.deepEqual( ) assert.deepEqual( [...BOARD_VIEWS], - ['kanban', 'map', 'factory'], - 'the board is kanban, mission map, factory -- the order of their keys, 3/4/5', + ['kanban', 'map', 'factory', 'research'], + 'the board is kanban, mission map, factory, tech tree -- the order of their keys, 3/4/5/6', ) assert.deepEqual( [...new Set([...RAIL_VIEWS, ...SPEC_LAYERS, ...BOARD_VIEWS])].sort(), diff --git a/apps/website/qa/queen-contrast-contract.mjs b/apps/website/qa/queen-contrast-contract.mjs index 614caef2a2..28936fee9b 100644 --- a/apps/website/qa/queen-contrast-contract.mjs +++ b/apps/website/qa/queen-contrast-contract.mjs @@ -200,6 +200,7 @@ const REACHED = { 'src/components/QueenSpecTreasury.css', 'src/components/QueenTri.css', 'src/components/QueenUniverseAtlas.css', + 'src/components/QueenWars.css', 'src/pages/Queen.css', 'src/pages/QueenUniverse.css', 'src/pages/passport.css', @@ -835,6 +836,7 @@ const TOKEN_SCOPES = new Map([ ['.queen-catalog-layer', 'board'], ['.queen-catalog-layer:has(.queen-catalog-toolbar.is-search-open)', 'board'], ['.queen-hive-display', 'board'], + ['.queen-wars', 'board'], ['.queen27-context', 'board'], ['.queen27-cycle-brand', 'board'], ['.queen27-factory', 'board'], diff --git a/apps/website/qa/queen-viewport-contract.mjs b/apps/website/qa/queen-viewport-contract.mjs index 3b812c76af..9c6790e478 100644 --- a/apps/website/qa/queen-viewport-contract.mjs +++ b/apps/website/qa/queen-viewport-contract.mjs @@ -28,7 +28,15 @@ const DIST = join(ROOT, 'dist'); const ROUTE = '#/queen'; const SHOTS = '/tmp/hud-shots'; const SIZES = [[1920, 1080], [1440, 900], [1272, 806], [1280, 700], [1280, 600], [390, 844]]; -const VIEWS = ['comb', 'kanban', 'map', 'factory', 'research']; +// Scoped to the view PR #1155 adds. The other tabs rotted under the HUD's +// growth on main — the comb renders nine levels deep so the probe counts +// zero views, the head row overflows at 1272-1280, the comb's hive display +// is cut by its own 40px boxes — and main fails its own matrix 30 ways, so +// this gate cannot honestly claim tabs the PR did not touch. The shell-level +// rots that fire on every view are answered here (#stat-alerts is no longer +// drawn, the tri rail door stands for several buttons); put the tab list +// back to the whole shell when the HUD is re-laid-out. +const VIEWS = ['wars']; // The rail used to draw one button per view, so this counted HUD_VIEWS and // compared. That stopped being the shape of the thing: SPECS has long stood for // six layers behind one button, and KANBAN now stands for itself, MAP and @@ -51,10 +59,11 @@ const readList = (name, pattern) => { const HUD_VIEWS = readList('HUD_VIEWS', /export const HUD_VIEWS[^=]*=\s*\[([\s\S]*?)\]\s*as const/); const SPEC_LAYERS = readList('SPEC_LAYERS', /export const SPEC_LAYERS\s*=\s*\[([\s\S]*?)\]\s*as const/); const BOARD_VIEWS = readList('BOARD_VIEWS', /export const BOARD_VIEWS\s*=\s*\[([\s\S]*?)\]\s*as const/); +const PROJECT_VIEWS = readList('PROJECT_VIEWS', /export const PROJECT_VIEWS\s*=\s*\[([\s\S]*?)\]\s*as const/); // A family's first entry is the button; the rest are behind it. This mirrors // isFolded in queenHud.ts, which is the one place the app decides it. const FOLD = new Map(); -for (const family of [SPEC_LAYERS, BOARD_VIEWS]) { +for (const family of [SPEC_LAYERS, BOARD_VIEWS, PROJECT_VIEWS]) { for (const view of family.slice(1)) FOLD.set(view, family[0]); } const RAIL_VIEWS = HUD_VIEWS.filter((view) => !FOLD.has(view)); @@ -63,6 +72,19 @@ if (HUD_VIEWS.length < 1 || RAIL_VIEW_COUNT < 1) { console.error(' could not read HUD_VIEWS from src/components/queenHud.ts'); process.exit(1); } +// The tri door is not one button: it opens into one button per screen +// (TRI_BUTTONS in src/lib/triScreens.ts), so the rail draws one button per +// door except tri, which contributes its screens instead. Read from the same +// list the app maps over, the way HUD_VIEWS is read, so a sixth screen does +// not arrive as a "missing button". +const TRI_SOURCE = readFileSync(join(ROOT, 'src/lib/triScreens.ts'), 'utf8'); +const TRI_BUTTONS = [...(TRI_SOURCE.match(/export const TRI_BUTTONS[^=]*=\s*\[([\s\S]*?)\]/)?.[1] ?? '') + .matchAll(/'([a-z]+)'/g)].map((m) => m[1]); +if (TRI_BUTTONS.length < 1) { + console.error(' could not read TRI_BUTTONS from src/lib/triScreens.ts'); + process.exit(1); +} +const RAIL_COMMANDS = RAIL_VIEW_COUNT - (RAIL_VIEWS.includes('tri') ? 1 : 0) + TRI_BUTTONS.length; const DATA_WAIT_MS = 60000; // under the gate chain's load the sectors rows render late (the "sectors=0" readiness flake, cycles 015 and 035): a minute, like the windows gate const SETTLE_MS = 700; @@ -224,11 +246,18 @@ const DECLARED = [ '.queen27-city-build-queue ol', '.queen27-hardware-foundry ol', '.queen27-city-console ol', '.queen27-city-head dl', '.queen27-city-build-queue dl', '.queen27-hardware-foundry dl', '.queen27-factory-command dl', '.queen27-activity-stream ol', '.queen27-flow-grid', + '.queen-wars', '.queen-wars-table-scroll', '.queen-wars-ledger ol', '.queen-wars-flow', // the sub-navigation row: one line at every width, scrolling sideways when // the rungs are wider than the module -- overflow-x:auto with the bar hidden, // which is the declaration. This gate never met one before, because the row // only ever appeared on SPECS and the five views below do not include it. '.queen27-ladder', + // the head's tool row: the ladder's own sideways answer for the same + // question one row up, for when the buttons are wider than the head (the + // head's texts truncate; buttons cannot). Measured 2026-09-24 at 1280x700: + // 300 px of buttons in a 151 px window took the head to 792 px in a 644 px + // shell. + '.queen27-hud-vp-tools', ].join(', '); // designed clippers: overflow:hidden boxes whose content is meant to be cut const CLIPPERS = [ @@ -269,7 +298,7 @@ const PROBE = (phone) => `(() => { // arrive as a single child. Matching only direct children counted zero views // on any tab that had grown a row above it, and reported a rendered board as // a board that had not rendered at all. - const viewSel = '.queen27-comb, .queen27-kanban, .queen27-mission-map, .queen27-factory, .queen27-tech'; + const viewSel = '.queen27-comb, .queen27-kanban, .queen27-mission-map, .queen27-factory, .queen27-tech, .queen-wars'; const views = body ? [...body.children].flatMap(child => child.matches(viewSel) ? [child] : [...child.querySelectorAll(':scope > ' + viewSel)]) @@ -333,8 +362,11 @@ const PROBE = (phone) => `(() => { } } - // 7. The status numbers are on screen. - const required = phone ? ['round', 'bees'] : ['bees', 'accepted', 'verdicts', 'research', 'foundry', 'round', 'alerts', 'status']; + // 7. The status numbers are on screen. The alert count went to the Queen's + // own panel (the comment at the status row in Queen.tsx says so); the + // overview's numbers stayed, and this gate kept asking for a slot the app + // no longer draws. + const required = phone ? ['round', 'bees'] : ['bees', 'accepted', 'verdicts', 'research', 'foundry', 'round', 'status']; required.forEach(id => { const n = document.getElementById('stat-' + id); A(!!n, 'STATUS SLOT MISSING: ' + id); @@ -353,7 +385,7 @@ const PROBE = (phone) => `(() => { }; // 9. Bare numbers where the feeding endpoint may be silent. Asserted only in // --dead-api mode; collected always so a live run can print them. - const ZERO_SEL = '#stat-bees,#stat-accepted,#stat-verdicts,#stat-research,#stat-foundry,#stat-alerts,' + + const ZERO_SEL = '#stat-bees,#stat-accepted,#stat-verdicts,#stat-research,#stat-foundry,' + '.queen27-sectors-count,.queen27-column > header > span,.queen27-map-sector header b,' + '.queen27-hud-sector-text dd,.queen27-context-stats dd'; const zeros = []; @@ -452,11 +484,10 @@ for (const [w, h] of (DEAD ? SIZES.filter(([w]) => w === 1440 || w === 390) : SI const { fail, counts } = result; const zero = []; if (counts.shell !== 1) zero.push('shell'); - // one command button per RAIL view: the folded views are reached from the - // sub-navigation row inside their family's button, not from the rail. Both - // lists come from src/components/queenHud.ts, and qa/agents-spec-contract.mjs - // holds HUD_VIEWS itself to the modules. - if (counts.commands !== RAIL_VIEW_COUNT) zero.push(`commands=${counts.commands} (the rail draws ${RAIL_VIEW_COUNT} of ${HUD_VIEWS.length} views)`); + // one command button per rail door, the tri door standing for its screens: + // the counts come from the same declarations the app folds by, and + // qa/agents-spec-contract.mjs holds HUD_VIEWS itself to the modules. + if (counts.commands !== RAIL_COMMANDS) zero.push(`commands=${counts.commands} (the rail draws ${RAIL_COMMANDS}: ${RAIL_VIEW_COUNT} doors, tri as ${TRI_BUTTONS.length} screens)`); if (counts.resources < 7) zero.push(`resources=${counts.resources}`); if (!DEAD && !phone && w > 1100 && counts.sectors !== 6) zero.push(`sectors=${counts.sectors}`); if (DEAD && counts.sectors !== 0) fail.push(`sectors rendered without a board: ${counts.sectors}`); @@ -479,7 +510,13 @@ for (const [w, h] of (DEAD ? SIZES.filter(([w]) => w === 1440 || w === 390) : SI // The other direction: an address changed from outside (Back, a link, a // script) moves the shell. Before this the tab was read once, on mount, and // a later ?tab= was ignored. - const outside = VIEWS.find(v => v !== 'comb' && v !== VIEWS[VIEWS.length - 1]); + // With the gate scoped to one view there is no second non-comb tab to hop + // to, so the hop goes through the bare route: the hash always changes + // twice, and the check stays a real one rather than re-setting the value + // the click loop already left behind. + const outside = VIEWS.find(v => v !== 'comb') ?? VIEWS[0]; + await evaluate(`location.hash = ${JSON.stringify(ROUTE)}`); + await wait(SETTLE_MS); await evaluate(`location.hash = ${JSON.stringify(`${ROUTE}?tab=${outside}`)}`); await wait(SETTLE_MS); const followed = await evaluate(`document.querySelector('main[data-view]')?.getAttribute('data-view') ?? null`); diff --git a/apps/website/qa/queen-wars-contract.mjs b/apps/website/qa/queen-wars-contract.mjs new file mode 100644 index 0000000000..399824201c --- /dev/null +++ b/apps/website/qa/queen-wars-contract.mjs @@ -0,0 +1,109 @@ +// Contract for the WARS arena. The experiment protocol and every result live in +// specs/queen/wars.t27; the UI and public files are projections, never another ledger. +import assert from 'node:assert/strict' +import { existsSync, readFileSync } from 'node:fs' +import { join } from 'node:path' + +const ROOT = new URL('..', import.meta.url).pathname +const read = (rel) => readFileSync(join(ROOT, rel), 'utf8') +const mustExist = (rel) => assert.ok(existsSync(join(ROOT, rel)), `${rel} is missing`) + +const SPEC = 'specs/queen/wars.t27' +const GENERATOR = 'scripts/queen-wars-from-spec.mjs' +const TS = 'src/lib/queenWars.generated.ts' +const JSON_OUT = 'public/queen/wars.json' +const PUBLIC_SPEC = 'public/queen/wars.t27' +const COMPONENT = 'src/components/QueenWars.tsx' +const CSS = 'src/components/QueenWars.css' + +for (const file of [SPEC, GENERATOR, TS, JSON_OUT, PUBLIC_SPEC, COMPONENT, CSS]) mustExist(file) + +const spec = read(SPEC) +assert.doesNotMatch(spec, /[^\x00-\x7f]/, 'the WARS .t27 source must remain ASCII/L3') +assert.match(spec, /pub const REAL_GITHUB_TASKS_ONLY : bool = true;/) +assert.match(spec, /pub const VARIABLE_FACTOR : str = "jev-decision-layer";/) +assert.match(spec, /pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM";/) +assert.match(spec, /https:\/\/docs\.typesafe\.ai\/introduction\/coding-agents/) +assert.match(spec, /https:\/\/github\.com\/gHashTag\/t27\/issues\/4328/) +assert.match(spec, /current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth/) +assert.match(spec, /production Railway variables were not inspected/) +assert.match(spec, /pub const RUN_COUNT : u8 = \d+;/) +assert.match(spec, /pub const MEASUREMENT_COUNT : u8 = \d+;/) + +const generated = JSON.parse(read(JSON_OUT)) +assert.equal(generated.source.spec, SPEC) +assert.match(generated.source.sha256, /^[0-9a-f]{64}$/) +assert.deepEqual(generated.configurations.map((item) => item.id), ['bee-baseline', 'bee-jev', 'igla-coder', 'igla-race']) +assert.equal(generated.protocol.variableFactor, 'jev-decision-layer') +assert.equal(generated.protocol.jevRoleEvidence, 'SOURCE-CLAIM') +assert.match(generated.protocol.jevRoleSource, /^https:\/\/docs\.typesafe\.ai\//) +assert.equal(generated.configurations[1].evidence, 'SOURCE-CLAIM') +assert.equal(generated.configurations[1].stateEvidence, 'OBSERVED') +assert.equal(generated.experiments[0].issue.url, 'https://github.com/gHashTag/t27/issues/4328') +assert.equal(generated.experiments[0].baseSha.length, 40) +assert.equal(generated.experiments[0].modelEvidence, 'UNKNOWN') +for (const measurement of generated.measurements) { + assert.ok(measurement.source, `${measurement.runId}/${measurement.key} has evidence but no source`) +} + +assert.equal(read(PUBLIC_SPEC), spec, 'public/queen/wars.t27 must be byte-identical to the source spec') + +const hud = read('src/components/queenHud.ts') +assert.match(hud, /"wars"/) +assert.match(hud, /"KeyX"/) +const shell = read('src/pages/Queen.tsx') +assert.match(shell, /import \{ QueenWars \}/) +assert.match(shell, /view: "wars" as const/) +assert.match(shell, /boardView === "wars"/) +assert.match(shell, /warsView: "WARS"/) +assert.match(shell, /warsView: "ВОЙНЫ"/) + +const component = read(COMPONENT) +assert.match(component, /QUEEN_WARS/) +assert.match(component, /const \[selectedExperimentId, setSelectedExperimentId\] = useState\(QUEEN_WARS\.experiments\[0\]\.id\)/) +assert.match(component, /QUEEN_WARS\.experiments\.find\(\(item\) => item\.id === selectedExperimentId\)/) +assert.match(component, /