diff --git a/.github/workflows/claude-code-review.yml b/.github/workflows/claude-code-review.yml index fc9448bceb..e4fe73f596 100644 --- a/.github/workflows/claude-code-review.yml +++ b/.github/workflows/claude-code-review.yml @@ -34,11 +34,44 @@ jobs: with: fetch-depth: 1 + # The review runs on z.ai (the owner, 2026-09-27: "замени ANTHROPIC_API_KEY + # на z.ai"). Every run before this failed two seconds in: the OAuth token + # was refused and ANTHROPIC_API_KEY was empty. z.ai serves the Anthropic + # Messages API at /api/anthropic, and Claude Code reaches it through + # ANTHROPIC_BASE_URL and ANTHROPIC_AUTH_TOKEN (docs.z.ai, the Claude Code + # guide). The key is the ZAI_API_KEY secret; a PR without it (a fork, or + # before the secret is set) is told so and not reviewed. + - name: Check for the z.ai key + id: zai + env: + ZAI_API_KEY: ${{ secrets.ZAI_API_KEY }} + run: | + if [ -n "$ZAI_API_KEY" ]; then + echo "ready=true" >> "$GITHUB_OUTPUT" + else + echo "ready=false" >> "$GITHUB_OUTPUT" + echo "::notice title=Review skipped::The ZAI_API_KEY secret is not set, so this PR is not reviewed. Add it under Settings > Secrets and variables > Actions." + fi + - name: Run Claude Code Review id: claude-review + if: steps.zai.outputs.ready == 'true' uses: anthropics/claude-code-action@v1 + env: + ANTHROPIC_BASE_URL: https://api.z.ai/api/anthropic + ANTHROPIC_AUTH_TOKEN: ${{ secrets.ZAI_API_KEY }} + # z.ai's guide raises Claude Code's request timeout for GLM. + API_TIMEOUT_MS: "3000000" + # Background and subagent calls ask for a Haiku, Sonnet or Opus by + # name; these send them to GLM instead. ZAI_MODEL (a repository + # variable) picks the model; glm-4.6 is the default because it is the + # z.ai id the Queen's own model catalogue lists. + ANTHROPIC_DEFAULT_OPUS_MODEL: ${{ vars.ZAI_MODEL || 'glm-4.6' }} + ANTHROPIC_DEFAULT_SONNET_MODEL: ${{ vars.ZAI_MODEL || 'glm-4.6' }} + ANTHROPIC_DEFAULT_HAIKU_MODEL: glm-4.5-air with: - claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} + # The action requires a key input; it is the same z.ai key. + anthropic_api_key: ${{ secrets.ZAI_API_KEY }} # Note: the log suggests `show_full_output: true`, but that is not an # input this action version accepts -- its inputs are trigger_phrase, # prompt, settings, claude_args and the auth/provider set. Extra @@ -60,5 +93,5 @@ jobs: # See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md # or https://code.claude.com/docs/en/cli-reference for available options - claude_args: '--verbose --allowed-tools "Bash(gh issue view:*),Bash(gh search:*),Bash(gh issue list:*),Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*),Bash(gh pr list:*)"' + claude_args: '--model ${{ vars.ZAI_MODEL || ''glm-4.6'' }} --verbose --allowed-tools "Bash(gh issue view:*),Bash(gh search:*),Bash(gh issue list:*),Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*),Bash(gh pr list:*)"' diff --git a/.github/workflows/website-checks.yml b/.github/workflows/website-checks.yml index 5336ebadc1..5820618cd0 100644 --- a/.github/workflows/website-checks.yml +++ b/.github/workflows/website-checks.yml @@ -271,6 +271,12 @@ jobs: - name: Queen WARS experiment ledger run: npm run check:wars && npm run test:wars-spec + # LEVEL II on the ROADMAP tab: the raid, honey, bosses, cracks and the + # round's pulse. The raid list is pinned here and in gHashTag/t27's + # feed_roadmap.py, so the page and the feeder raid the same sector. + - name: Roadmap game rules + run: npm run check:roadmap-game + - name: Build run: npx vite build diff --git a/apps/website/package.json b/apps/website/package.json index 7e443d7e1a..ba12e7aec8 100644 --- a/apps/website/package.json +++ b/apps/website/package.json @@ -97,6 +97,7 @@ "check:onboarding": "node scripts/onboarding-from-spec.mjs --check", "check:wars": "node scripts/queen-wars-from-spec.mjs --check && node --experimental-strip-types qa/queen-wars-contract.mjs", "test:wars-spec": "node --test scripts/queen-wars-from-spec.test.mjs", + "check:roadmap-game": "node --experimental-strip-types qa/roadmap-game-contract.mjs", "test:viewport-spec": "node --test scripts/viewport-from-spec.test.mjs", "check:explorer-viewport": "node qa/explorer-viewport-contract.mjs", "check:passport": "node --experimental-strip-types qa/passport-figures.mjs" diff --git a/apps/website/public/queen/runs/campaign-20260926/base-4613.txt b/apps/website/public/queen/runs/campaign-20260926/base-4613.txt new file mode 100644 index 0000000000..e8c66a1d31 --- /dev/null +++ b/apps/website/public/queen/runs/campaign-20260926/base-4613.txt @@ -0,0 +1,41 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4613 +# worktree: /home/user/ghashtag/t27 +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:59:37Z +# diff --numstat: (no change) + +## criterion 1: FAIL (expected 1) +$ t27c coverage specs/base/ternary_encoding.t27 2>&1 | grep -cE '^Untested: +0$' +0 + +## criterion 2: PASS (expected 13) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/base/ternary_encoding.t27 +13 + +## criterion 3: FAIL (expected >= 11) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/base/ternary_encoding.t27 +10 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/base/ternary_encoding.t27 +IMPLEMENTED + +## review FR-002 parse +$ t27c parse specs/base/ternary_encoding.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/base --top 500 2>&1 | grep -F ternary_encoding.t27 || echo "ternary_encoding.t27: not listed as vacuous" +ternary_encoding.t27: not listed as vacuous + +## review test-report +$ t27c test-report specs/base/ternary_encoding.t27 2>&1 | tail -15 +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:280:47: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +# ACCEPTANCE: FAILED diff --git a/apps/website/public/queen/runs/campaign-20260926/base-4614.txt b/apps/website/public/queen/runs/campaign-20260926/base-4614.txt new file mode 100644 index 0000000000..cefa7da354 --- /dev/null +++ b/apps/website/public/queen/runs/campaign-20260926/base-4614.txt @@ -0,0 +1,46 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4614 +# worktree: /home/user/ghashtag/t27 +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:51:51Z +# diff --numstat: (no change) + +## criterion 1: FAIL (expected 1) +$ t27c coverage specs/boards/arty_a7.t27 2>&1 | grep -cE '^Untested: +0$' +0 + +## criterion 2: PASS (expected 5) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/boards/arty_a7.t27 +5 + +## criterion 3: FAIL (expected >= 18) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/boards/arty_a7.t27 +17 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/boards/arty_a7.t27 +IMPLEMENTED + +## criterion 5: PASS (expected 0) +$ t27c test-report specs/boards/arty_a7.t27 2>&1 | grep -c BLOCKED +0 + +## review FR-002 parse +$ t27c parse specs/boards/arty_a7.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/boards --top 500 2>&1 | grep -F arty_a7.t27 || echo "arty_a7.t27: not listed as vacuous" +specs/boards/arty_a7.t27 0 0 0.0% 0 0 + +## review test-report +$ t27c test-report specs/boards/arty_a7.t27 2>&1 | tail -15 +--- test report: specs/boards/arty_a7.t27 --- + + tests 17 + pass 17 + FAIL 0 + invariants 11 proved -- comptime, so compiling IS the check + rate 100.0% + +# ACCEPTANCE: FAILED diff --git a/apps/website/public/queen/runs/campaign-20260926/base-4695.txt b/apps/website/public/queen/runs/campaign-20260926/base-4695.txt new file mode 100644 index 0000000000..c59b79e3a8 --- /dev/null +++ b/apps/website/public/queen/runs/campaign-20260926/base-4695.txt @@ -0,0 +1,46 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4695 +# worktree: /home/user/ghashtag/t27 +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:59:36Z +# diff --numstat: (no change) + +## criterion 1: FAIL (expected 1) +$ t27c coverage specs/fpga/testbench/simulator_tb.t27 2>&1 | grep -cE '^Untested: +0$' +0 + +## criterion 2: PASS (expected 4) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/fpga/testbench/simulator_tb.t27 +4 + +## criterion 3: FAIL (expected >= 7) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/fpga/testbench/simulator_tb.t27 +6 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/fpga/testbench/simulator_tb.t27 +IMPLEMENTED + +## criterion 5: PASS (expected 0) +$ t27c test-report specs/fpga/testbench/simulator_tb.t27 2>&1 | grep -c BLOCKED +0 + +## review FR-002 parse +$ t27c parse specs/fpga/testbench/simulator_tb.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/fpga/testbench --top 500 2>&1 | grep -F simulator_tb.t27 || echo "simulator_tb.t27: not listed as vacuous" +specs/fpga/testbench/simulator_tb.t27 7 0 0.0% 1 0 + +## review test-report +$ t27c test-report specs/fpga/testbench/simulator_tb.t27 2>&1 | tail -15 +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + + tests 6 + pass 6 + FAIL 0 + invariants 1 proved -- comptime, so compiling IS the check + rate 100.0% + +# ACCEPTANCE: FAILED diff --git a/apps/website/public/queen/runs/campaign-20260926/judge/accept.py b/apps/website/public/queen/runs/campaign-20260926/judge/accept.py new file mode 100644 index 0000000000..fab52bf9a6 --- /dev/null +++ b/apps/website/public/queen/runs/campaign-20260926/judge/accept.py @@ -0,0 +1,72 @@ +#!/usr/bin/env python3 +"""Queen WARS acceptance judge: run an issue's own acceptance commands in a worktree. + +Usage: accept.py + +Runs every acceptance criterion exactly as the issue states it, from the +worktree root, with the judge t27c and zig 0.16.0 on PATH, and writes a +transcript: command, raw output, expected value, pass/fail. Then two review +checks the issue's requirements name (FR-002 parse, vacuity) and the spec's +own test-report summary. Exit 0 iff every acceptance criterion passes. +""" +import datetime +import os +import subprocess +import sys + +JUDGE_PATH = "/home/user/wars/judge:/home/user/tools/bin" + +SPECS = { + "4614": ("specs/boards/arty_a7.t27", 5, 18, True), + "4695": ("specs/fpga/testbench/simulator_tb.t27", 4, 7, True), + "4613": ("specs/base/ternary_encoding.t27", 13, 11, False), +} + + +def sh(cmd, cwd): + env = dict(os.environ, PATH=JUDGE_PATH + ":" + os.environ["PATH"]) + p = subprocess.run(["bash", "-c", cmd], cwd=cwd, env=env, + capture_output=True, text=True, timeout=900) + return (p.stdout + p.stderr).strip() + + +def main(): + issue, wt, out = sys.argv[1], sys.argv[2], sys.argv[3] + spec, n_fn, min_tests, has_c5 = SPECS[issue] + crit = [ + ("1", f"t27c coverage {spec} 2>&1 | grep -cE '^Untested: +0$'", lambda o: o == "1", "1"), + ("2", f"grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' {spec}", + lambda o: o == str(n_fn), str(n_fn)), + ("3", f"grep -cE '^[[:space:]]*test[[:space:]]+(\"|[A-Za-z_])' {spec}", + lambda o: o.isdigit() and int(o) >= min_tests, f">= {min_tests}"), + ("4", f"t27c spec-status {spec}", lambda o: "IMPLEMENTED" in o.split(), "IMPLEMENTED"), + ] + if has_c5: + crit.append(("5", f"t27c test-report {spec} 2>&1 | grep -c BLOCKED", + lambda o: o == "0", "0")) + lines = [f"# Queen WARS acceptance transcript: gHashTag/t27#{issue}", + f"# worktree: {wt}", + f"# head: {sh('git rev-parse HEAD', wt)}", + f"# judge: {sh('t27c --version', wt)}; zig {sh('zig version', wt)}", + f"# at: {datetime.datetime.now(datetime.timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')}", + f"# diff --numstat: {sh('git diff --numstat', wt) or '(no change)'}", + ""] + ok_all = True + for n, cmd, ok, want in crit: + o = sh(cmd, wt) + passed = ok(o) + ok_all &= passed + lines += [f"## criterion {n}: {'PASS' if passed else 'FAIL'} (expected {want})", + f"$ {cmd}", o, ""] + for label, cmd in (("review FR-002 parse", f"t27c parse {spec} >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err"), + ("review vacuity", f"t27c validate-vacuity --specs-dir {os.path.dirname(spec)} --top 500 2>&1 | grep -F {os.path.basename(spec)} || echo \"{os.path.basename(spec)}: not listed as vacuous\""), + ("review test-report", f"t27c test-report {spec} 2>&1 | tail -15")): + lines += [f"## {label}", f"$ {cmd}", sh(cmd, wt), ""] + lines.append(f"# ACCEPTANCE: {'PASSED' if ok_all else 'FAILED'}") + open(out, "w").write("\n".join(lines) + "\n") + print(f"#{issue}: {'PASSED' if ok_all else 'FAILED'} -> {out}") + return 0 if ok_all else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/apps/website/public/queen/runs/campaign-20260926/judge/mutate.py b/apps/website/public/queen/runs/campaign-20260926/judge/mutate.py new file mode 100644 index 0000000000..ba3e1dc116 --- /dev/null +++ b/apps/website/public/queen/runs/campaign-20260926/judge/mutate.py @@ -0,0 +1,114 @@ +#!/usr/bin/env python3 +"""Queen WARS review: does an arm's NEW test catch a fixed set of mutants? + +Usage: mutate.py + +For each mutant of the function under test (fixed per issue, the same for +every arm), rewrite the arm's patched spec, run `t27c test-report --verbose`, +and record whether the arm's new test FAILs (killed), passes (survived) or +the spec does not build (blocked). The arm's file is restored byte-for-byte +after every mutant, and the restore is checked. +""" +import os +import subprocess +import sys + +PATH = "/home/user/wars/judge:/home/user/tools/bin:" + os.environ["PATH"] + +MUTANTS = { + "4614": ("specs/boards/arty_a7.t27", [ + ("return NUM_BUTTONS + 1", "fn count_buttons() -> usize {\n return NUM_BUTTONS;", + "fn count_buttons() -> usize {\n return NUM_BUTTONS + 1;"), + ("return 0", "fn count_buttons() -> usize {\n return NUM_BUTTONS;", + "fn count_buttons() -> usize {\n return 0;"), + ("return NUM_LEDS - 1", "fn count_buttons() -> usize {\n return NUM_BUTTONS;", + "fn count_buttons() -> usize {\n return NUM_LEDS - 1;"), + ]), + "4695": ("specs/fpga/testbench/simulator_tb.t27", [ + ("cycle += 2", " sim_cycle = sim_cycle + 1;\n }\n\n fn reset", + " sim_cycle = sim_cycle + 2;\n }\n\n fn reset"), + ("clock ends low", " clk = false;\n clk = true;\n sim_cycle = sim_cycle + 1;", + " clk = true;\n clk = false;\n sim_cycle = sim_cycle + 1;"), + ("no cycle advance", " sim_cycle = sim_cycle + 1;\n }\n\n fn reset", + " sim_cycle = sim_cycle;\n }\n\n fn reset"), + ("tick also counts an event", " sim_cycle = sim_cycle + 1;\n }\n\n fn reset", + " sim_cycle = sim_cycle + 1;\n events_processed = events_processed + 1;\n }\n\n fn reset"), + ]), + "4613": ("specs/base/ternary_encoding.t27", [ + ("never rejects", " if (!is_valid_trit(trits[i])) {\n return false;", + " if (false) {\n return false;"), + ("checks only the first slot", "fn validate_trits(trits: []i32, len: usize) → bool {\n var i : usize = 0;\n while (i < len) {", + "fn validate_trits(trits: []i32, len: usize) → bool {\n var i : usize = 0;\n while (i < 1) {"), + ("skips the last slot", "fn validate_trits(trits: []i32, len: usize) → bool {\n var i : usize = 0;\n while (i < len) {", + "fn validate_trits(trits: []i32, len: usize) → bool {\n var i : usize = 0;\n while (i + 1 < len) {"), + ("rejects everything", " i = i + 1;\n }\n return true;\n }\n\n // ═", + " i = i + 1;\n }\n return false;\n }\n\n // ═"), + ]), +} + +# Pre-existing blockers in ternary_encoding.t27 at afe2186c, outside #4613's +# boundary: two comptime invariants that fail (balanced encoders, unipolar +# decoders) and two tests that discard a return value. Removed only in the +# review copy, so an arm's new test can be run at all. +DEBLOCK = {"4613": ["invariant byte_trits_roundtrip", "invariant bits_trits_roundtrip", + "test byte_to_trits_roundtrip", "test char_encoding_roundtrip"]} + + +def deblock(text, names): + out, skip = [], False + for line in text.split("\n"): + head = line[4:] if line.startswith(" ") and not line.startswith(" ") else None + if head is not None or line.startswith("}"): + skip = head is not None and any(head == n or head.startswith(n + " ") + for n in names) + if not skip: + out.append(line) + return "\n".join(out) + + +def main(): + issue, wt, test_name, out = sys.argv[1:5] + source = sys.argv[5] if len(sys.argv) > 5 else None # arm file to review in wt + spec, mutants = MUTANTS[issue] + path = os.path.join(wt, spec) + if source: + text = deblock(open(source).read(), DEBLOCK[issue]) + open(path, "w").write(text) + original = open(path, "rb").read() + lines, killed = [f"# mutation review: gHashTag/t27#{issue}, new test `{test_name}`" + + (f" (de-blocked copy of {source})" if source else ""), ""], 0 + if source: + r = subprocess.run(["bash", "-c", f"t27c test-report --verbose {spec} 2>&1"], + cwd=wt, env=dict(os.environ, PATH=PATH), + capture_output=True, text=True, timeout=900) + lines += ["## unmutated de-blocked copy", r.stdout.strip(), ""] + for label, old, new in mutants: + text = original.decode() + if text.count(old) != 1: + lines.append(f"## {label}: SKIPPED (anchor not unique: {text.count(old)})") + continue + try: + open(path, "w").write(text.replace(old, new)) + r = subprocess.run(["bash", "-c", f"t27c test-report --verbose {spec} 2>&1"], + cwd=wt, env=dict(os.environ, PATH=PATH), + capture_output=True, text=True, timeout=900) + rep = r.stdout + finally: + open(path, "wb").write(original) + assert open(path, "rb").read() == original, "restore failed" + if "BLOCKED" in rep: + verdict = "blocked" + elif f"FAIL {test_name}" in rep: + verdict, killed = "killed", killed + 1 + elif f"pass {test_name}" in rep: + verdict = "survived" + else: + verdict = "not-run" + lines += [f"## {label}: {verdict}", rep.strip(), ""] + lines.append(f"# MUTANTS KILLED BY THE NEW TEST: {killed}/{len(mutants)}") + open(out, "w").write("\n".join(lines) + "\n") + print(f"#{issue} {os.path.basename(wt)}: {killed}/{len(mutants)} killed -> {out}") + + +if __name__ == "__main__": + main() diff --git a/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/acceptance.txt b/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/acceptance.txt new file mode 100644 index 0000000000..63a4c28be2 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/acceptance.txt @@ -0,0 +1,41 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4613 +# worktree: /home/user/wars/wt/4613-baseline +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:57:06Z +# diff --numstat: 12 0 specs/base/ternary_encoding.t27 + +## criterion 1: PASS (expected 1) +$ t27c coverage specs/base/ternary_encoding.t27 2>&1 | grep -cE '^Untested: +0$' +1 + +## criterion 2: PASS (expected 13) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/base/ternary_encoding.t27 +13 + +## criterion 3: PASS (expected >= 11) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/base/ternary_encoding.t27 +11 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/base/ternary_encoding.t27 +IMPLEMENTED + +## review FR-002 parse +$ t27c parse specs/base/ternary_encoding.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/base --top 500 2>&1 | grep -F ternary_encoding.t27 || echo "ternary_encoding.t27: not listed as vacuous" +ternary_encoding.t27: not listed as vacuous + +## review test-report +$ t27c test-report specs/base/ternary_encoding.t27 2>&1 | tail -15 +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:293:47: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +# ACCEPTANCE: PASSED diff --git a/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/mutants.txt b/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/mutants.txt new file mode 100644 index 0000000000..3d78405665 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/mutants.txt @@ -0,0 +1,73 @@ +# mutation review: gHashTag/t27#4613, new test `validate_trits_check` (de-blocked copy of wt/4613-baseline/specs/base/ternary_encoding.t27) + +## unmutated de-blocked copy +--- test report: specs/base/ternary_encoding.t27 --- + pass bit_to_trit_pair_zero + pass bit_to_trit_pair_one + pass bits_to_trits_zero + FAIL bits_to_trits_max_nibble + FAIL trits_to_bits_roundtrip + pass balanced_unipolar_conversion + pass is_valid_trit_check + pass is_valid_unipolar_trit_check + pass validate_trits_check + + tests 9 + pass 7 + FAIL 2 + invariants 3 proved -- comptime, so compiling IS the check + rate 77.8% + +## never rejects: blocked +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:254:61: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +## checks only the first slot: killed +--- test report: specs/base/ternary_encoding.t27 --- + pass bit_to_trit_pair_zero + pass bit_to_trit_pair_one + pass bits_to_trits_zero + FAIL bits_to_trits_max_nibble + FAIL trits_to_bits_roundtrip + pass balanced_unipolar_conversion + pass is_valid_trit_check + pass is_valid_unipolar_trit_check + FAIL validate_trits_check + + tests 9 + pass 6 + FAIL 3 + invariants 3 proved -- comptime, so compiling IS the check + rate 66.7% + +## skips the last slot: killed +--- test report: specs/base/ternary_encoding.t27 --- + pass bit_to_trit_pair_zero + pass bit_to_trit_pair_one + pass bits_to_trits_zero + FAIL bits_to_trits_max_nibble + FAIL trits_to_bits_roundtrip + pass balanced_unipolar_conversion + pass is_valid_trit_check + pass is_valid_unipolar_trit_check + FAIL validate_trits_check + + tests 9 + pass 6 + FAIL 3 + invariants 3 proved -- comptime, so compiling IS the check + rate 66.7% + +## rejects everything: blocked +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:252:58: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +# MUTANTS KILLED BY THE NEW TEST: 2/4 diff --git a/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/patch.diff b/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/patch.diff new file mode 100644 index 0000000000..2a059b5401 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z/patch.diff @@ -0,0 +1,23 @@ +diff --git a/specs/base/ternary_encoding.t27 b/specs/base/ternary_encoding.t27 +index 8e9c72e..77f0295 100644 +--- a/specs/base/ternary_encoding.t27 ++++ b/specs/base/ternary_encoding.t27 +@@ -308,6 +308,18 @@ module TernaryEncoding { + assert is_valid_unipolar_trit(3) == false + assert is_valid_unipolar_trit(-1) == false + ++ // validate_trits: true iff every trit in the first len slots is balanced; ++ // slots at or past len are not inspected, so an empty prefix is valid ++ test validate_trits_check ++ var all_valid : [3]i32 = [_]i32{TRIT_NEG, TRIT_ZERO, TRIT_POS}; ++ var bad_last : [3]i32 = [_]i32{TRIT_POS, TRIT_ZERO, 2}; ++ var bad_first : [3]i32 = [_]i32{-2, TRIT_ZERO, TRIT_POS}; ++ assert validate_trits(&all_valid, 3) == true ++ assert validate_trits(&bad_last, 3) == false ++ assert validate_trits(&bad_last, 2) == true ++ assert validate_trits(&bad_first, 3) == false ++ assert validate_trits(&bad_first, 0) == true ++ + test char_encoding_roundtrip + var test_chars : [5]u8 = [_]u8{'A', '0', ' ', '\n', 127}; + var trits : [10]i32 = undefined; diff --git a/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/acceptance.txt b/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/acceptance.txt new file mode 100644 index 0000000000..f74c06a514 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/acceptance.txt @@ -0,0 +1,41 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4613 +# worktree: /home/user/wars/wt/4613-tri +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:56:44Z +# diff --numstat: 18 0 specs/base/ternary_encoding.t27 + +## criterion 1: PASS (expected 1) +$ t27c coverage specs/base/ternary_encoding.t27 2>&1 | grep -cE '^Untested: +0$' +1 + +## criterion 2: PASS (expected 13) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/base/ternary_encoding.t27 +13 + +## criterion 3: PASS (expected >= 11) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/base/ternary_encoding.t27 +11 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/base/ternary_encoding.t27 +IMPLEMENTED + +## review FR-002 parse +$ t27c parse specs/base/ternary_encoding.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/base --top 500 2>&1 | grep -F ternary_encoding.t27 || echo "ternary_encoding.t27: not listed as vacuous" +ternary_encoding.t27: not listed as vacuous + +## review test-report +$ t27c test-report specs/base/ternary_encoding.t27 2>&1 | tail -15 +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:294:47: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +# ACCEPTANCE: PASSED diff --git a/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/mutants.txt b/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/mutants.txt new file mode 100644 index 0000000000..a34249ac4b --- /dev/null +++ b/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/mutants.txt @@ -0,0 +1,73 @@ +# mutation review: gHashTag/t27#4613, new test `validate_trits_check` (de-blocked copy of wt/4613-tri/specs/base/ternary_encoding.t27) + +## unmutated de-blocked copy +--- test report: specs/base/ternary_encoding.t27 --- + pass bit_to_trit_pair_zero + pass bit_to_trit_pair_one + pass bits_to_trits_zero + FAIL bits_to_trits_max_nibble + FAIL trits_to_bits_roundtrip + pass balanced_unipolar_conversion + pass is_valid_trit_check + pass is_valid_unipolar_trit_check + pass validate_trits_check + + tests 9 + pass 7 + FAIL 2 + invariants 3 proved -- comptime, so compiling IS the check + rate 77.8% + +## never rejects: blocked +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:255:61: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +## checks only the first slot: killed +--- test report: specs/base/ternary_encoding.t27 --- + pass bit_to_trit_pair_zero + pass bit_to_trit_pair_one + pass bits_to_trits_zero + FAIL bits_to_trits_max_nibble + FAIL trits_to_bits_roundtrip + pass balanced_unipolar_conversion + pass is_valid_trit_check + pass is_valid_unipolar_trit_check + FAIL validate_trits_check + + tests 9 + pass 6 + FAIL 3 + invariants 3 proved -- comptime, so compiling IS the check + rate 66.7% + +## skips the last slot: killed +--- test report: specs/base/ternary_encoding.t27 --- + pass bit_to_trit_pair_zero + pass bit_to_trit_pair_one + pass bits_to_trits_zero + FAIL bits_to_trits_max_nibble + FAIL trits_to_bits_roundtrip + pass balanced_unipolar_conversion + pass is_valid_trit_check + pass is_valid_unipolar_trit_check + FAIL validate_trits_check + + tests 9 + pass 6 + FAIL 3 + invariants 3 proved -- comptime, so compiling IS the check + rate 66.7% + +## rejects everything: blocked +--- test report: specs/base/ternary_encoding.t27 --- + BLOCKED does not compile: /tmp/t27c-test-report-ternary_encoding/spec.zig:8:9: error: assertion failed + /tmp/t27c-test-report-ternary_encoding/spec.zig:253:58: note: called at comptime here + + A blocked spec is not a failing one. It never produced a + binary, so it has no per-test result to report. + +# MUTANTS KILLED BY THE NEW TEST: 2/4 diff --git a/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/patch.diff b/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/patch.diff new file mode 100644 index 0000000000..879a023f00 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z/patch.diff @@ -0,0 +1,29 @@ +diff --git a/specs/base/ternary_encoding.t27 b/specs/base/ternary_encoding.t27 +index 8e9c72e..cc76fc3 100644 +--- a/specs/base/ternary_encoding.t27 ++++ b/specs/base/ternary_encoding.t27 +@@ -308,6 +308,24 @@ module TernaryEncoding { + assert is_valid_unipolar_trit(3) == false + assert is_valid_unipolar_trit(-1) == false + ++ test validate_trits_check ++ // Every balanced trit value is accepted ++ var balanced : [5]i32 = [_]i32{TRIT_NEG, TRIT_ZERO, TRIT_POS, TRIT_ZERO, TRIT_NEG}; ++ assert validate_trits(&balanced, 5) == true ++ ++ // A unipolar digit 2 is not a balanced trit, even in the last slot ++ var unipolar : [3]i32 = [_]i32{TRIT_ZERO, TRIT_POS, 2}; ++ assert validate_trits(&unipolar, 3) == false ++ ++ // Out of range below TRIT_NEG is rejected ++ var too_low : [4]i32 = [_]i32{TRIT_POS, -2, TRIT_ZERO, TRIT_NEG}; ++ assert validate_trits(&too_low, 4) == false ++ ++ // Only the first len trits are checked ++ assert validate_trits(&unipolar, 2) == true ++ assert validate_trits(&too_low, 1) == true ++ assert validate_trits(&too_low, 0) == true ++ + test char_encoding_roundtrip + var test_chars : [5]u8 = [_]u8{'A', '0', ' ', '\n', 127}; + var trits : [10]i32 = undefined; diff --git a/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/acceptance.txt b/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/acceptance.txt new file mode 100644 index 0000000000..4f55ef60a5 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/acceptance.txt @@ -0,0 +1,46 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4614 +# worktree: /home/user/wars/wt/4614-baseline +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:53:50Z +# diff --numstat: 4 0 specs/boards/arty_a7.t27 + +## criterion 1: PASS (expected 1) +$ t27c coverage specs/boards/arty_a7.t27 2>&1 | grep -cE '^Untested: +0$' +1 + +## criterion 2: PASS (expected 5) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/boards/arty_a7.t27 +5 + +## criterion 3: PASS (expected >= 18) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/boards/arty_a7.t27 +18 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/boards/arty_a7.t27 +IMPLEMENTED + +## criterion 5: PASS (expected 0) +$ t27c test-report specs/boards/arty_a7.t27 2>&1 | grep -c BLOCKED +0 + +## review FR-002 parse +$ t27c parse specs/boards/arty_a7.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/boards --top 500 2>&1 | grep -F arty_a7.t27 || echo "arty_a7.t27: not listed as vacuous" +specs/boards/arty_a7.t27 0 0 0.0% 0 0 + +## review test-report +$ t27c test-report specs/boards/arty_a7.t27 2>&1 | tail -15 +--- test report: specs/boards/arty_a7.t27 --- + + tests 18 + pass 18 + FAIL 0 + invariants 11 proved -- comptime, so compiling IS the check + rate 100.0% + +# ACCEPTANCE: PASSED diff --git a/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/mutants.txt b/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/mutants.txt new file mode 100644 index 0000000000..20ee4fb424 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/mutants.txt @@ -0,0 +1,84 @@ +# mutation review: gHashTag/t27#4614, new test `count_buttons_is_4` + +## return NUM_BUTTONS + 1: killed +--- test report: specs/boards/arty_a7.t27 --- + pass board_name_set + pass fpga_family_artix7 + pass clock_freq_100mhz + pass clock_period_10ns + pass num_leds_is_4 + pass num_buttons_is_4 + FAIL count_buttons_is_4 + pass has_uart_true + pass has_spi_true + pass clock_freq_mhz_100 + pass pin_clk_is_e3 + pass pin_rst_n_is_c12 + pass pin_uart_tx_is_a9 + pass pin_uart_rx_is_c9 + pass pin_led0_is_r5 + pass pin_led3_is_t9 + pass pin_btn0_is_d9 + pass two_fpga_parts_available + + tests 18 + pass 17 + FAIL 1 + invariants 11 proved -- comptime, so compiling IS the check + rate 94.4% + +## return 0: killed +--- test report: specs/boards/arty_a7.t27 --- + pass board_name_set + pass fpga_family_artix7 + pass clock_freq_100mhz + pass clock_period_10ns + pass num_leds_is_4 + pass num_buttons_is_4 + FAIL count_buttons_is_4 + pass has_uart_true + pass has_spi_true + pass clock_freq_mhz_100 + pass pin_clk_is_e3 + pass pin_rst_n_is_c12 + pass pin_uart_tx_is_a9 + pass pin_uart_rx_is_c9 + pass pin_led0_is_r5 + pass pin_led3_is_t9 + pass pin_btn0_is_d9 + pass two_fpga_parts_available + + tests 18 + pass 17 + FAIL 1 + invariants 11 proved -- comptime, so compiling IS the check + rate 94.4% + +## return NUM_LEDS - 1: killed +--- test report: specs/boards/arty_a7.t27 --- + pass board_name_set + pass fpga_family_artix7 + pass clock_freq_100mhz + pass clock_period_10ns + pass num_leds_is_4 + pass num_buttons_is_4 + FAIL count_buttons_is_4 + pass has_uart_true + pass has_spi_true + pass clock_freq_mhz_100 + pass pin_clk_is_e3 + pass pin_rst_n_is_c12 + pass pin_uart_tx_is_a9 + pass pin_uart_rx_is_c9 + pass pin_led0_is_r5 + pass pin_led3_is_t9 + pass pin_btn0_is_d9 + pass two_fpga_parts_available + + tests 18 + pass 17 + FAIL 1 + invariants 11 proved -- comptime, so compiling IS the check + rate 94.4% + +# MUTANTS KILLED BY THE NEW TEST: 3/3 diff --git a/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/patch.diff b/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/patch.diff new file mode 100644 index 0000000000..0ead35fbf2 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z/patch.diff @@ -0,0 +1,15 @@ +diff --git a/specs/boards/arty_a7.t27 b/specs/boards/arty_a7.t27 +index ed25cdd..0307fb4 100644 +--- a/specs/boards/arty_a7.t27 ++++ b/specs/boards/arty_a7.t27 +@@ -197,6 +197,10 @@ module BoardArtyA7 { + given n = NUM_BUTTONS + then n == 4 + ++ test count_buttons_is_4 ++ given n = count_buttons() ++ then n == NUM_BUTTONS and n == 4 ++ + test has_uart_true + given result = has_uart() + then result == true diff --git a/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/acceptance.txt b/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/acceptance.txt new file mode 100644 index 0000000000..7a8eeaa992 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/acceptance.txt @@ -0,0 +1,46 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4614 +# worktree: /home/user/wars/wt/4614-tri +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:53:58Z +# diff --numstat: 4 0 specs/boards/arty_a7.t27 + +## criterion 1: PASS (expected 1) +$ t27c coverage specs/boards/arty_a7.t27 2>&1 | grep -cE '^Untested: +0$' +1 + +## criterion 2: PASS (expected 5) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/boards/arty_a7.t27 +5 + +## criterion 3: PASS (expected >= 18) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/boards/arty_a7.t27 +18 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/boards/arty_a7.t27 +IMPLEMENTED + +## criterion 5: PASS (expected 0) +$ t27c test-report specs/boards/arty_a7.t27 2>&1 | grep -c BLOCKED +0 + +## review FR-002 parse +$ t27c parse specs/boards/arty_a7.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/boards --top 500 2>&1 | grep -F arty_a7.t27 || echo "arty_a7.t27: not listed as vacuous" +specs/boards/arty_a7.t27 0 0 0.0% 0 0 + +## review test-report +$ t27c test-report specs/boards/arty_a7.t27 2>&1 | tail -15 +--- test report: specs/boards/arty_a7.t27 --- + + tests 18 + pass 18 + FAIL 0 + invariants 11 proved -- comptime, so compiling IS the check + rate 100.0% + +# ACCEPTANCE: PASSED diff --git a/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/mutants.txt b/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/mutants.txt new file mode 100644 index 0000000000..20ee4fb424 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/mutants.txt @@ -0,0 +1,84 @@ +# mutation review: gHashTag/t27#4614, new test `count_buttons_is_4` + +## return NUM_BUTTONS + 1: killed +--- test report: specs/boards/arty_a7.t27 --- + pass board_name_set + pass fpga_family_artix7 + pass clock_freq_100mhz + pass clock_period_10ns + pass num_leds_is_4 + pass num_buttons_is_4 + FAIL count_buttons_is_4 + pass has_uart_true + pass has_spi_true + pass clock_freq_mhz_100 + pass pin_clk_is_e3 + pass pin_rst_n_is_c12 + pass pin_uart_tx_is_a9 + pass pin_uart_rx_is_c9 + pass pin_led0_is_r5 + pass pin_led3_is_t9 + pass pin_btn0_is_d9 + pass two_fpga_parts_available + + tests 18 + pass 17 + FAIL 1 + invariants 11 proved -- comptime, so compiling IS the check + rate 94.4% + +## return 0: killed +--- test report: specs/boards/arty_a7.t27 --- + pass board_name_set + pass fpga_family_artix7 + pass clock_freq_100mhz + pass clock_period_10ns + pass num_leds_is_4 + pass num_buttons_is_4 + FAIL count_buttons_is_4 + pass has_uart_true + pass has_spi_true + pass clock_freq_mhz_100 + pass pin_clk_is_e3 + pass pin_rst_n_is_c12 + pass pin_uart_tx_is_a9 + pass pin_uart_rx_is_c9 + pass pin_led0_is_r5 + pass pin_led3_is_t9 + pass pin_btn0_is_d9 + pass two_fpga_parts_available + + tests 18 + pass 17 + FAIL 1 + invariants 11 proved -- comptime, so compiling IS the check + rate 94.4% + +## return NUM_LEDS - 1: killed +--- test report: specs/boards/arty_a7.t27 --- + pass board_name_set + pass fpga_family_artix7 + pass clock_freq_100mhz + pass clock_period_10ns + pass num_leds_is_4 + pass num_buttons_is_4 + FAIL count_buttons_is_4 + pass has_uart_true + pass has_spi_true + pass clock_freq_mhz_100 + pass pin_clk_is_e3 + pass pin_rst_n_is_c12 + pass pin_uart_tx_is_a9 + pass pin_uart_rx_is_c9 + pass pin_led0_is_r5 + pass pin_led3_is_t9 + pass pin_btn0_is_d9 + pass two_fpga_parts_available + + tests 18 + pass 17 + FAIL 1 + invariants 11 proved -- comptime, so compiling IS the check + rate 94.4% + +# MUTANTS KILLED BY THE NEW TEST: 3/3 diff --git a/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/patch.diff b/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/patch.diff new file mode 100644 index 0000000000..0ead35fbf2 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z/patch.diff @@ -0,0 +1,15 @@ +diff --git a/specs/boards/arty_a7.t27 b/specs/boards/arty_a7.t27 +index ed25cdd..0307fb4 100644 +--- a/specs/boards/arty_a7.t27 ++++ b/specs/boards/arty_a7.t27 +@@ -197,6 +197,10 @@ module BoardArtyA7 { + given n = NUM_BUTTONS + then n == 4 + ++ test count_buttons_is_4 ++ given n = count_buttons() ++ then n == NUM_BUTTONS and n == 4 ++ + test has_uart_true + given result = has_uart() + then result == true diff --git a/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/acceptance.txt b/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/acceptance.txt new file mode 100644 index 0000000000..a5ff6a4e06 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/acceptance.txt @@ -0,0 +1,46 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4695 +# worktree: /home/user/wars/wt/4695-baseline +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:55:22Z +# diff --numstat: 16 0 specs/fpga/testbench/simulator_tb.t27 + +## criterion 1: PASS (expected 1) +$ t27c coverage specs/fpga/testbench/simulator_tb.t27 2>&1 | grep -cE '^Untested: +0$' +1 + +## criterion 2: PASS (expected 4) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/fpga/testbench/simulator_tb.t27 +4 + +## criterion 3: PASS (expected >= 7) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/fpga/testbench/simulator_tb.t27 +7 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/fpga/testbench/simulator_tb.t27 +IMPLEMENTED + +## criterion 5: PASS (expected 0) +$ t27c test-report specs/fpga/testbench/simulator_tb.t27 2>&1 | grep -c BLOCKED +0 + +## review FR-002 parse +$ t27c parse specs/fpga/testbench/simulator_tb.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/fpga/testbench --top 500 2>&1 | grep -F simulator_tb.t27 || echo "simulator_tb.t27: not listed as vacuous" +specs/fpga/testbench/simulator_tb.t27 8 0 0.0% 1 0 + +## review test-report +$ t27c test-report specs/fpga/testbench/simulator_tb.t27 2>&1 | tail -15 +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + + tests 7 + pass 7 + FAIL 0 + invariants 1 proved -- comptime, so compiling IS the check + rate 100.0% + +# ACCEPTANCE: PASSED diff --git a/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/mutants.txt b/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/mutants.txt new file mode 100644 index 0000000000..bc874f652b --- /dev/null +++ b/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/mutants.txt @@ -0,0 +1,67 @@ +# mutation review: gHashTag/t27#4695, new test `test_tick_advances_one_cycle` + +## cycle += 2: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick_advances_one_cycle + pass test_reset_state + FAIL test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 5 + FAIL 2 + invariants 1 proved -- comptime, so compiling IS the check + rate 71.4% + +## clock ends low: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick_advances_one_cycle + pass test_reset_state + pass test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 6 + FAIL 1 + invariants 1 proved -- comptime, so compiling IS the check + rate 85.7% + +## no cycle advance: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick_advances_one_cycle + pass test_reset_state + FAIL test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 5 + FAIL 2 + invariants 1 proved -- comptime, so compiling IS the check + rate 71.4% + +## tick also counts an event: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick_advances_one_cycle + pass test_reset_state + FAIL test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 5 + FAIL 2 + invariants 1 proved -- comptime, so compiling IS the check + rate 71.4% + +# MUTANTS KILLED BY THE NEW TEST: 4/4 diff --git a/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/patch.diff b/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/patch.diff new file mode 100644 index 0000000000..eb70a2b278 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z/patch.diff @@ -0,0 +1,27 @@ +diff --git a/specs/fpga/testbench/simulator_tb.t27 b/specs/fpga/testbench/simulator_tb.t27 +index d16d81d..fe30722 100644 +--- a/specs/fpga/testbench/simulator_tb.t27 ++++ b/specs/fpga/testbench/simulator_tb.t27 +@@ -51,6 +51,22 @@ module Simulator_Testbench { + return true; + } + ++ // tick() is one clock cycle: the clock ends high, the cycle counter ++ // advances by exactly one, and nothing else moves -- event counting ++ // belongs to run_cycles(), and reset is left released. ++ test test_tick_advances_one_cycle { ++ reset(); ++ var events_before : u32 = events_processed; ++ tick(); ++ invariant sim_cycle == 1; ++ invariant clk == true; ++ tick(); ++ invariant sim_cycle == 2; ++ invariant clk == true; ++ invariant rst_n == true; ++ invariant events_processed == events_before; ++ } ++ + test test_reset_state { + reset(); + invariant sim_running == false; diff --git a/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/acceptance.txt b/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/acceptance.txt new file mode 100644 index 0000000000..f3ad1fd075 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/acceptance.txt @@ -0,0 +1,46 @@ +# Queen WARS acceptance transcript: gHashTag/t27#4695 +# worktree: /home/user/wars/wt/4695-tri +# head: afe2186cfa9572c8c2d519506aa85a8c9447eae5 +# judge: t27c 0.4.0; zig 0.16.0 +# at: 2026-09-26T18:55:23Z +# diff --numstat: 16 0 specs/fpga/testbench/simulator_tb.t27 + +## criterion 1: PASS (expected 1) +$ t27c coverage specs/fpga/testbench/simulator_tb.t27 2>&1 | grep -cE '^Untested: +0$' +1 + +## criterion 2: PASS (expected 4) +$ grep -cE '^[[:space:]]*(pub[[:space:]]+)?fn[[:space:]]' specs/fpga/testbench/simulator_tb.t27 +4 + +## criterion 3: PASS (expected >= 7) +$ grep -cE '^[[:space:]]*test[[:space:]]+("|[A-Za-z_])' specs/fpga/testbench/simulator_tb.t27 +7 + +## criterion 4: PASS (expected IMPLEMENTED) +$ t27c spec-status specs/fpga/testbench/simulator_tb.t27 +IMPLEMENTED + +## criterion 5: PASS (expected 0) +$ t27c test-report specs/fpga/testbench/simulator_tb.t27 2>&1 | grep -c BLOCKED +0 + +## review FR-002 parse +$ t27c parse specs/fpga/testbench/simulator_tb.t27 >/dev/null 2>/tmp/t27c_parse_err; echo exit=$?; head -3 /tmp/t27c_parse_err +exit=0 + +## review vacuity +$ t27c validate-vacuity --specs-dir specs/fpga/testbench --top 500 2>&1 | grep -F simulator_tb.t27 || echo "simulator_tb.t27: not listed as vacuous" +specs/fpga/testbench/simulator_tb.t27 8 0 0.0% 1 0 + +## review test-report +$ t27c test-report specs/fpga/testbench/simulator_tb.t27 2>&1 | tail -15 +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + + tests 7 + pass 7 + FAIL 0 + invariants 1 proved -- comptime, so compiling IS the check + rate 100.0% + +# ACCEPTANCE: PASSED diff --git a/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/mutants.txt b/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/mutants.txt new file mode 100644 index 0000000000..2fa1a8ece5 --- /dev/null +++ b/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/mutants.txt @@ -0,0 +1,67 @@ +# mutation review: gHashTag/t27#4695, new test `test_tick` + +## cycle += 2: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick + pass test_reset_state + FAIL test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 5 + FAIL 2 + invariants 1 proved -- comptime, so compiling IS the check + rate 71.4% + +## clock ends low: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick + pass test_reset_state + pass test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 6 + FAIL 1 + invariants 1 proved -- comptime, so compiling IS the check + rate 85.7% + +## no cycle advance: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick + pass test_reset_state + FAIL test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 5 + FAIL 2 + invariants 1 proved -- comptime, so compiling IS the check + rate 71.4% + +## tick also counts an event: killed +--- test report: specs/fpga/testbench/simulator_tb.t27 --- + FAIL test_tick + pass test_reset_state + FAIL test_run_cycles + pass test_schedule_event + pass test_schedule_overflow + pass test_max_cycles + pass test_event_queue_depth + + tests 7 + pass 5 + FAIL 2 + invariants 1 proved -- comptime, so compiling IS the check + rate 71.4% + +# MUTANTS KILLED BY THE NEW TEST: 4/4 diff --git a/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/patch.diff b/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/patch.diff new file mode 100644 index 0000000000..01d6eab21c --- /dev/null +++ b/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z/patch.diff @@ -0,0 +1,27 @@ +diff --git a/specs/fpga/testbench/simulator_tb.t27 b/specs/fpga/testbench/simulator_tb.t27 +index d16d81d..f7abe3b 100644 +--- a/specs/fpga/testbench/simulator_tb.t27 ++++ b/specs/fpga/testbench/simulator_tb.t27 +@@ -51,6 +51,22 @@ module Simulator_Testbench { + return true; + } + ++ // tick() drives one clock edge (clk ends high) and advances the cycle ++ // counter by exactly one; counting events is run_cycles' job, not tick's. ++ // Measured relative to the state on entry, so the test holds whatever ++ // ran before it. No reset() first: it would leave clk already high. ++ test test_tick { ++ var cycle_before : u32 = sim_cycle; ++ var events_before : u32 = events_processed; ++ tick(); ++ invariant clk == true; ++ invariant sim_cycle == cycle_before + 1; ++ tick(); ++ invariant clk == true; ++ invariant sim_cycle == cycle_before + 2; ++ invariant events_processed == events_before; ++ } ++ + test test_reset_state { + reset(); + invariant sim_running == false; diff --git a/apps/website/public/queen/wars.json b/apps/website/public/queen/wars.json index 308cf1bf46..7b7a567bcf 100644 --- a/apps/website/public/queen/wars.json +++ b/apps/website/public/queen/wars.json @@ -2,7 +2,7 @@ "source": { "spec": "specs/queen/wars.t27", "publicSpec": "public/queen/wars.t27", - "sha256": "fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8", + "sha256": "43e610413821968b743749cb01bed33a54eb9c3b05203b16035c2be2128da5d6", "schemaVersion": 2 }, "name": "Queen WARS real-task agent arena", @@ -24,6 +24,7 @@ "planned", "running", "credential-blocked", + "judged", "complete", "invalid" ], @@ -107,6 +108,18 @@ { "key": "jev-confidence", "unit": "probability" + }, + { + "key": "total-tokens", + "unit": "tokens" + }, + { + "key": "judge-calls", + "unit": "count" + }, + { + "key": "mutants-killed", + "unit": "count" } ], "configurations": [ @@ -116,12 +129,24 @@ "kind": "coding-agent", "state": "ready", "evidence": "OBSERVED", - "source": "current Codex task runtime", - "note": "Control arm: the coding Bee chooses and implements without an external decision model.", + "source": "coding agent runtime: Codex task runtime (2026-09-23 run); Claude Code subagents (2026-09-26 campaign)", + "note": "Control arm: the coding Bee chooses and implements without an external decision layer. In the 2026-09-26 campaign it had no t27c while it worked; the gates judged it afterwards.", "stateEvidence": "OBSERVED", - "stateSource": "current Codex task runtime on 2026-09-23", + "stateSource": "current Codex task runtime on 2026-09-23; Claude Code subagents on 2026-09-26", "stateNote": "The baseline Bee can run locally." }, + { + "id": "bee-tri", + "name": "Bee + TRI", + "kind": "coding-agent-plus-compiler-judge", + "state": "ready", + "evidence": "OBSERVED", + "source": "t27c 0.4.0 built from the gHashTag/t27 bootstrap at afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "note": "Variable arm: the same Bee, prompt, tools and budget, plus the t27c compiler as its decision layer. It may run read-only t27c subcommands to choose between candidate edits; t27c never writes the patch.", + "stateEvidence": "OBSERVED", + "stateSource": "Queen WARS campaign 2026-09-26: six paired arms on gHashTag/t27 issues 4614, 4695 and 4613", + "stateNote": "Runs wherever t27c and zig 0.16.0 build; no credential is needed." + }, { "id": "bee-jev", "name": "Bee + JEV", @@ -129,7 +154,7 @@ "state": "credential-blocked", "evidence": "SOURCE-CLAIM", "source": "https://docs.typesafe.ai/introduction/coding-agents", - "note": "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", + "note": "Comparison arm only: the same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "stateEvidence": "OBSERVED", "stateSource": "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "stateNote": "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed." @@ -142,9 +167,9 @@ "evidence": "OBSERVED", "source": "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "note": "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", - "stateEvidence": "OBSERVED", - "stateSource": "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", - "stateNote": "No executable checkpoint was verified for this arena." + "stateEvidence": "SOURCE-CLAIM", + "stateSource": "gHashTag/igla-coder-gpu LEDGER.md and c_infer/model.bin.json, read 2026-09-26; the repository is private", + "stateNote": "The IGLA pilot reports trained checkpoints (tern_tc: 9.08M parameters, 0.7613 bits per byte after 2.0B tokens; a 100M full-precision and ternary pair) and 2.0% and 1.3% pass@10 on a t27 fill-in-the-body bench. The checkpoints are not public, and none has run an arena issue." }, { "id": "igla-race", @@ -183,6 +208,78 @@ "state": "credential-blocked", "evidence": "OBSERVED", "note": "This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed." + }, + { + "id": "t27-4614-count-buttons", + "name": "count_buttons missing test", + "issue": { + "repo": "gHashTag/t27", + "number": 4614, + "url": "https://github.com/gHashTag/t27/issues/4614", + "updatedAt": "2026-09-23T09:31:53Z" + }, + "baseSha": "afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "executorModel": "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", + "prompt": "Resolve gHashTag/t27 issue 4614 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4614.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", + "promptSha256": "76e1a46c458f60b3ff900d49b4ac8d22a209c6aa19948573951c4b5fcf75e953", + "toolPolicy": "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", + "toolPolicySha256": "803db5972a1ba27b9139c51f5790e6f477ff1065b36fbc06331a2b166d6ea41a", + "budget": "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", + "acceptance": "t27c coverage specs/boards/arty_a7.t27 reports Untested: 0; function count remains 5; test count is at least 18; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", + "acceptanceSha256": "b677c416dec6736730e8b730528ca8a98649780f808f8c6e64eb548133202d59", + "modelEvidence": "SESSION-OBSERVED", + "modelSource": "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", + "state": "judged", + "evidence": "OBSERVED", + "note": "Both arms wrote the same 4-line test, byte for byte, and both kill 3 of 3 mutants of count_buttons. There is no difference to rank." + }, + { + "id": "t27-4695-tick", + "name": "tick missing test", + "issue": { + "repo": "gHashTag/t27", + "number": 4695, + "url": "https://github.com/gHashTag/t27/issues/4695", + "updatedAt": "2026-09-24T02:32:52Z" + }, + "baseSha": "afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "executorModel": "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", + "prompt": "Resolve gHashTag/t27 issue 4695 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4695.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", + "promptSha256": "720c9cfcc51101e421d3941e0840f30484d2700aa71aa405a7b7707e7a89c2e5", + "toolPolicy": "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", + "toolPolicySha256": "803db5972a1ba27b9139c51f5790e6f477ff1065b36fbc06331a2b166d6ea41a", + "budget": "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", + "acceptance": "t27c coverage specs/fpga/testbench/simulator_tb.t27 reports Untested: 0; function count remains 4; test count is at least 7; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", + "acceptanceSha256": "cd26ee3c8267d84e28d55cbd0753c2112d10b650555d48893f670a6dddd564ee", + "modelEvidence": "SESSION-OBSERVED", + "modelSource": "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", + "state": "judged", + "evidence": "OBSERVED", + "note": "Different tests; both are accepted and both kill 4 of 4 mutants of tick. The TRI arm also found a compiler defect: an assignment to a module var inside a test body is lowered as a new local, so the generated Zig fails with a shadowing error." + }, + { + "id": "t27-4613-validate-trits", + "name": "validate_trits missing test (re-filed 4328)", + "issue": { + "repo": "gHashTag/t27", + "number": 4613, + "url": "https://github.com/gHashTag/t27/issues/4613", + "updatedAt": "2026-09-23T09:31:51Z" + }, + "baseSha": "afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "executorModel": "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", + "prompt": "Resolve gHashTag/t27 issue 4613 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4613.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", + "promptSha256": "a444d68feffff727cea821062fdaa1ff836c2a565858a29715de8f27bb4b5a25", + "toolPolicy": "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", + "toolPolicySha256": "803db5972a1ba27b9139c51f5790e6f477ff1065b36fbc06331a2b166d6ea41a", + "budget": "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", + "acceptance": "t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; the issue has no test-report criterion", + "acceptanceSha256": "a9378e97de7a7765226678447f2da1fc8b63c0b70531f8c1def9e6c82c52d07b", + "modelEvidence": "SESSION-OBSERVED", + "modelSource": "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", + "state": "judged", + "evidence": "OBSERVED", + "note": "Both arms pass the four criteria, but the spec is BLOCKED at the base and this issue has no test-report criterion, so the new test cannot run in the repository. In a review copy with the four pre-existing blockers removed, both tests pass and kill 2 of 4 mutants; the other 2 are caught at compile time by an existing invariant. The TRI arm traced the blocker to balanced-trit encoders paired with unipolar decoders." } ], "runs": [ @@ -199,6 +296,90 @@ "patchSha256": "11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267", "artifactUrl": null, "note": "The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed." + }, + { + "id": "t27-4614-bee-baseline-20260926T185257Z", + "experimentId": "t27-4614-count-buttons", + "configId": "bee-baseline", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:54:00Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "6561a0821694e2a4afa55ddcfe9c4d960223c77487737f0d7e636c713e5f3950", + "patchSha256": "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z", + "note": "Added test count_buttons_is_4 without a compiler; every criterion passed when judged." + }, + { + "id": "t27-4614-bee-tri-20260926T185257Z", + "experimentId": "t27-4614-count-buttons", + "configId": "bee-tri", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:54:02Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "a4db3eb859fb458f0ba512b119ac41c4e350d2d111597927d5f032b80e9ca096", + "patchSha256": "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z", + "note": "Same patch as the baseline, byte for byte; also ran a mutation check of its own before reporting." + }, + { + "id": "t27-4695-bee-baseline-20260926T185257Z", + "experimentId": "t27-4695-tick", + "configId": "bee-baseline", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:55:26Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "444d27dc87851aaaf5ba7a642df4f6838bbcba043ab99c08dfc47baea5318f24", + "patchSha256": "cf7115ad1f649cc302646c3713488f21d810c47c3f01e1d01a7d72021a889f6b", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z", + "note": "Added test_tick_advances_one_cycle after reset(), argued from the compiler source; every criterion passed when judged." + }, + { + "id": "t27-4695-bee-tri-20260926T185257Z", + "experimentId": "t27-4695-tick", + "configId": "bee-tri", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:55:26Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "4a8ea205b62bede5c36157fe7dc3c7536966c0f6580398f1ce104aef3dd55dc3", + "patchSha256": "aa0443796cf2e7a0ca10dd1b5f885883e6ecaff97d5be77d50a994983e81b7ce", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z", + "note": "Added test_tick relative to the entry state; found that assigning a module var in a test body generates a shadowing local." + }, + { + "id": "t27-4613-bee-baseline-20260926T185257Z", + "experimentId": "t27-4613-validate-trits", + "configId": "bee-baseline", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:57:10Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "25689dbaf0221824f23f6bcb7ec0f8556fec2cf40832ab226c68333d817b3805", + "patchSha256": "6f44011b4ff970ffd2536c18baf755b64ea69f01539bde11243bcc1fcc743e96", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z", + "note": "Added validate_trits_check with five asserts; the four criteria pass, and test-report stays BLOCKED as at the base." + }, + { + "id": "t27-4613-bee-tri-20260926T185257Z", + "experimentId": "t27-4613-validate-trits", + "configId": "bee-tri", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:56:37Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "48716f028e82cc6bc95fe51e3c46de1803a47f2480292ab76c9356d86337dd7a", + "patchSha256": "6c3e1562d398c7e59aacb44b09216463456f295354163590154d42279ded636a", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z", + "note": "Added validate_trits_check with six asserts; traced the pre-existing BLOCKED to mismatched trit encoders and decoders." } ], "measurements": [ @@ -241,6 +422,390 @@ "unit": "lines", "evidence": "SESSION-OBSERVED", "source": "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "elapsed", + "value": "62703", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "tool-calls", + "value": "10", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "patch-lines", + "value": "4", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "total-tokens", + "value": "76852", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "judge-calls", + "value": "0", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "mutants-killed", + "value": "3", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "elapsed", + "value": "64182", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "tool-calls", + "value": "14", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "patch-lines", + "value": "4", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "total-tokens", + "value": "77744", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "judge-calls", + "value": "8", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "mutants-killed", + "value": "3", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "elapsed", + "value": "148105", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "tool-calls", + "value": "30", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "patch-lines", + "value": "16", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "total-tokens", + "value": "100044", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "judge-calls", + "value": "0", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "mutants-killed", + "value": "4", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "elapsed", + "value": "148367", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "tool-calls", + "value": "27", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "patch-lines", + "value": "16", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "total-tokens", + "value": "98507", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "judge-calls", + "value": "9", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "mutants-killed", + "value": "4", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "elapsed", + "value": "252696", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "tool-calls", + "value": "38", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "patch-lines", + "value": "12", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 12 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "total-tokens", + "value": "128055", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "judge-calls", + "value": "0", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "mutants-killed", + "value": "2", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "elapsed", + "value": "219864", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "tool-calls", + "value": "27", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "patch-lines", + "value": "18", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 18 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "total-tokens", + "value": "113494", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "judge-calls", + "value": "17", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "mutants-killed", + "value": "2", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile" } ] } diff --git a/apps/website/public/queen/wars.t27 b/apps/website/public/queen/wars.t27 index a49983c9fb..3fa6db22f3 100644 --- a/apps/website/public/queen/wars.t27 +++ b/apps/website/public/queen/wars.t27 @@ -16,9 +16,11 @@ pub const SCHEMA_VERSION : u8 = 2; pub const GENERATED : [3]str = ["src/lib/queenWars.generated.ts", "public/queen/wars.json", "public/queen/wars.t27"]; // Epistemic and lifecycle vocabularies. Every displayed claim uses one of these. +// An experiment is "judged" when every arm ran and the acceptance gates judged it, +// but no winner may be declared (for example, the executor model id is not recorded). pub const EVIDENCE_LEVELS : [5]str = ["OBSERVED", "SESSION-OBSERVED", "SOURCE-CLAIM", "TARGET", "UNKNOWN"]; pub const CONFIG_STATES : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; -pub const EXPERIMENT_STATES : [5]str = ["planned", "running", "credential-blocked", "complete", "invalid"]; +pub const EXPERIMENT_STATES : [6]str = ["planned", "running", "credential-blocked", "judged", "complete", "invalid"]; pub const RUN_STATES : [5]str = ["pending", "running", "passed", "failed", "blocked"]; pub const VERDICTS : [4]str = ["accepted", "rejected", "inconclusive", "not-reviewed"]; pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; @@ -38,67 +40,69 @@ pub const TRI_ROLE_SOURCE : str = "https://t27.ai/t27/files/trinity/apps/website // Measurement vocabulary. Values are strings so an exact native unit is preserved; // absent rows mean unknown. A numeric zero is a measured zero only when a source exists. -pub const METRIC_COUNT : u8 = 11; -pub const METRIC_KEYS : [11]str = ["acceptance", "queen-verdict", "elapsed", "input-tokens", "output-tokens", "tool-calls", "retries", "patch-lines", "cost", "jev-latency", "jev-confidence"]; -pub const METRIC_UNITS : [11]str = ["verdict", "verdict", "ms", "tokens", "tokens", "count", "count", "lines", "usd", "ms", "probability"]; +pub const METRIC_COUNT : u8 = 14; +pub const METRIC_KEYS : [14]str = ["acceptance", "queen-verdict", "elapsed", "input-tokens", "output-tokens", "tool-calls", "retries", "patch-lines", "cost", "jev-latency", "jev-confidence", "total-tokens", "judge-calls", "mutants-killed"]; +pub const METRIC_UNITS : [14]str = ["verdict", "verdict", "ms", "tokens", "tokens", "count", "count", "lines", "usd", "ms", "probability", "tokens", "count", "count"]; -// Arena configurations. IGLA entries are visible now, but neither is presented as a -// measured coding model until an executable checkpoint and a witnessed run exist. -pub const CONFIG_COUNT : u8 = 4; -pub const CONFIG_IDS : [4]str = ["bee-baseline", "bee-jev", "igla-coder", "igla-race"]; -pub const CONFIG_NAMES : [4]str = ["Bee baseline", "Bee + JEV", "IGLA CODER", "IGLA RACE"]; -pub const CONFIG_KINDS : [4]str = ["coding-agent", "coding-agent-plus-decision-layer", "model-training-target", "training-race-pipeline"]; -pub const CONFIG_STATES_BY_ID : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; -pub const CONFIG_EVIDENCE : [4]str = ["OBSERVED", "SOURCE-CLAIM", "OBSERVED", "OBSERVED"]; -pub const CONFIG_SOURCES : [4]str = ["current Codex task runtime", "https://docs.typesafe.ai/introduction/coding-agents", "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "https://github.com/gHashTag/trios-trainer-igla"]; -pub const CONFIG_NOTES : [4]str = ["Control arm: the coding Bee chooses and implements without an external decision model.", "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result."]; -pub const CONFIG_STATE_EVIDENCE : [4]str = ["OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; -pub const CONFIG_STATE_SOURCES : [4]str = ["current Codex task runtime on 2026-09-23", "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", "repository inspection of the linked IGLA RACE pipeline on 2026-09-23"]; -pub const CONFIG_STATE_NOTES : [4]str = ["The baseline Bee can run locally.", "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed.", "No executable checkpoint was verified for this arena.", "The training pipeline is visible, but no coding-model run is claimed."]; +// Arena configurations. bee-baseline and bee-tri are the paired arms: the only +// difference is the TRI decision layer (t27c available while the Bee works). JEV +// stays as a comparison arm. IGLA entries are visible now, but neither is presented +// as a measured coding model until an executable checkpoint and a witnessed run exist. +pub const CONFIG_COUNT : u8 = 5; +pub const CONFIG_IDS : [5]str = ["bee-baseline", "bee-tri", "bee-jev", "igla-coder", "igla-race"]; +pub const CONFIG_NAMES : [5]str = ["Bee baseline", "Bee + TRI", "Bee + JEV", "IGLA CODER", "IGLA RACE"]; +pub const CONFIG_KINDS : [5]str = ["coding-agent", "coding-agent-plus-compiler-judge", "coding-agent-plus-decision-layer", "model-training-target", "training-race-pipeline"]; +pub const CONFIG_STATES_BY_ID : [5]str = ["ready", "ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; +pub const CONFIG_EVIDENCE : [5]str = ["OBSERVED", "OBSERVED", "SOURCE-CLAIM", "OBSERVED", "OBSERVED"]; +pub const CONFIG_SOURCES : [5]str = ["coding agent runtime: Codex task runtime (2026-09-23 run); Claude Code subagents (2026-09-26 campaign)", "t27c 0.4.0 built from the gHashTag/t27 bootstrap at afe2186cfa9572c8c2d519506aa85a8c9447eae5", "https://docs.typesafe.ai/introduction/coding-agents", "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "https://github.com/gHashTag/trios-trainer-igla"]; +pub const CONFIG_NOTES : [5]str = ["Control arm: the coding Bee chooses and implements without an external decision layer. In the 2026-09-26 campaign it had no t27c while it worked; the gates judged it afterwards.", "Variable arm: the same Bee, prompt, tools and budget, plus the t27c compiler as its decision layer. It may run read-only t27c subcommands to choose between candidate edits; t27c never writes the patch.", "Comparison arm only: the same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result."]; +pub const CONFIG_STATE_EVIDENCE : [5]str = ["OBSERVED", "OBSERVED", "OBSERVED", "SOURCE-CLAIM", "OBSERVED"]; +pub const CONFIG_STATE_SOURCES : [5]str = ["current Codex task runtime on 2026-09-23; Claude Code subagents on 2026-09-26", "Queen WARS campaign 2026-09-26: six paired arms on gHashTag/t27 issues 4614, 4695 and 4613", "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "gHashTag/igla-coder-gpu LEDGER.md and c_infer/model.bin.json, read 2026-09-26; the repository is private", "repository inspection of the linked IGLA RACE pipeline on 2026-09-23"]; +pub const CONFIG_STATE_NOTES : [5]str = ["The baseline Bee can run locally.", "Runs wherever t27c and zig 0.16.0 build; no credential is needed.", "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed.", "The IGLA pilot reports trained checkpoints (tern_tc: 9.08M parameters, 0.7613 bits per byte after 2.0B tokens; a 100M full-precision and ternary pair) and 2.0% and 1.3% pass@10 on a t27 fill-in-the-body bench. The checkpoints are not public, and none has run an arena issue.", "The training pipeline is visible, but no coding-model run is claimed."]; // Experiments. The exact prompt, tools, budget and acceptance live here, not in React. -pub const EXPERIMENT_COUNT : u8 = 1; -pub const EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; -pub const EXPERIMENT_NAMES : [1]str = ["validate_trits missing test"]; -pub const EXPERIMENT_REPOS : [1]str = ["gHashTag/t27"]; -pub const EXPERIMENT_ISSUE_URLS : [1]str = ["https://github.com/gHashTag/t27/issues/4328"]; -pub const EXPERIMENT_ISSUE_NUMBERS : [1]str = ["4328"]; -pub const EXPERIMENT_ISSUE_UPDATED_AT : [1]str = ["2026-09-20T13:51:01Z"]; -pub const EXPERIMENT_BASE_SHAS : [1]str = ["f123674fe40d6ba600a8c5c4948683198f756fdc"]; -pub const EXPERIMENT_EXECUTOR_MODELS : [1]str = ["UNKNOWN: exact model id and reasoning configuration were not exposed"]; -pub const EXPERIMENT_MODEL_EVIDENCE : [1]str = ["UNKNOWN"]; -pub const EXPERIMENT_MODEL_SOURCES : [1]str = ["current Codex task runtime did not expose a stable model id and reasoning configuration"]; -pub const EXPERIMENT_PROMPTS : [1]str = ["Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push."]; -pub const EXPERIMENT_TOOL_POLICIES : [1]str = ["Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push."]; -pub const EXPERIMENT_BUDGETS : [1]str = ["one Codex agent turn; no explicit token cap; record usage only when exposed"]; -pub const EXPERIMENT_ACCEPTANCE : [1]str = ["t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass"]; -pub const EXPERIMENT_STATES_BY_ID : [1]str = ["credential-blocked"]; -pub const EXPERIMENT_EVIDENCE : [1]str = ["OBSERVED"]; -pub const EXPERIMENT_NOTES : [1]str = ["This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed."]; +pub const EXPERIMENT_COUNT : u8 = 4; +pub const EXPERIMENT_IDS : [4]str = ["t27-4328-validate-trits", "t27-4614-count-buttons", "t27-4695-tick", "t27-4613-validate-trits"]; +pub const EXPERIMENT_NAMES : [4]str = ["validate_trits missing test", "count_buttons missing test", "tick missing test", "validate_trits missing test (re-filed 4328)"]; +pub const EXPERIMENT_REPOS : [4]str = ["gHashTag/t27", "gHashTag/t27", "gHashTag/t27", "gHashTag/t27"]; +pub const EXPERIMENT_ISSUE_URLS : [4]str = ["https://github.com/gHashTag/t27/issues/4328", "https://github.com/gHashTag/t27/issues/4614", "https://github.com/gHashTag/t27/issues/4695", "https://github.com/gHashTag/t27/issues/4613"]; +pub const EXPERIMENT_ISSUE_NUMBERS : [4]str = ["4328", "4614", "4695", "4613"]; +pub const EXPERIMENT_ISSUE_UPDATED_AT : [4]str = ["2026-09-20T13:51:01Z", "2026-09-23T09:31:53Z", "2026-09-24T02:32:52Z", "2026-09-23T09:31:51Z"]; +pub const EXPERIMENT_BASE_SHAS : [4]str = ["f123674fe40d6ba600a8c5c4948683198f756fdc", "afe2186cfa9572c8c2d519506aa85a8c9447eae5", "afe2186cfa9572c8c2d519506aa85a8c9447eae5", "afe2186cfa9572c8c2d519506aa85a8c9447eae5"]; +pub const EXPERIMENT_EXECUTOR_MODELS : [4]str = ["UNKNOWN: exact model id and reasoning configuration were not exposed", "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository"]; +pub const EXPERIMENT_MODEL_EVIDENCE : [4]str = ["UNKNOWN", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED"]; +pub const EXPERIMENT_MODEL_SOURCES : [4]str = ["current Codex task runtime did not expose a stable model id and reasoning configuration", "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it"]; +pub const EXPERIMENT_PROMPTS : [4]str = ["Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push.", "Resolve gHashTag/t27 issue 4614 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4614.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", "Resolve gHashTag/t27 issue 4695 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4695.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", "Resolve gHashTag/t27 issue 4613 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4613.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED."]; +pub const EXPERIMENT_TOOL_POLICIES : [4]str = ["Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push.", "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms."]; +pub const EXPERIMENT_BUDGETS : [4]str = ["one Codex agent turn; no explicit token cap; record usage only when exposed", "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime"]; +pub const EXPERIMENT_ACCEPTANCE : [4]str = ["t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass", "t27c coverage specs/boards/arty_a7.t27 reports Untested: 0; function count remains 5; test count is at least 18; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", "t27c coverage specs/fpga/testbench/simulator_tb.t27 reports Untested: 0; function count remains 4; test count is at least 7; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", "t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; the issue has no test-report criterion"]; +pub const EXPERIMENT_STATES_BY_ID : [4]str = ["credential-blocked", "judged", "judged", "judged"]; +pub const EXPERIMENT_EVIDENCE : [4]str = ["OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const EXPERIMENT_NOTES : [4]str = ["This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed.", "Both arms wrote the same 4-line test, byte for byte, and both kill 3 of 3 mutants of count_buttons. There is no difference to rank.", "Different tests; both are accepted and both kill 4 of 4 mutants of tick. The TRI arm also found a compiler defect: an assignment to a module var inside a test body is lowered as a new local, so the generated Zig fails with a shadowing error.", "Both arms pass the four criteria, but the spec is BLOCKED at the base and this issue has no test-report criterion, so the new test cannot run in the repository. In a review copy with the four pre-existing blockers removed, both tests pass and kill 2 of 4 mutants; the other 2 are caught at compile time by an existing invariant. The TRI arm traced the blocker to balanced-trit encoders paired with unipolar decoders."]; // Append-only run ledger. Completed, failed and blocked attempts all stay visible. -pub const RUN_COUNT : u8 = 1; -pub const RUN_IDS : [1]str = ["t27-4328-bee-baseline-20260923T041025Z"]; -pub const RUN_EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; -pub const RUN_CONFIG_IDS : [1]str = ["bee-baseline"]; -pub const RUN_STARTED_AT : [1]str = ["2026-09-23T04:10:25Z"]; -pub const RUN_FINISHED_AT : [1]str = ["2026-09-23T04:32:27Z"]; -pub const RUN_STATES_BY_ID : [1]str = ["blocked"]; -pub const RUN_EVIDENCE : [1]str = ["SESSION-OBSERVED"]; -pub const RUN_VERDICTS : [1]str = ["inconclusive"]; -pub const RUN_LOG_SHAS : [1]str = [""]; -pub const RUN_PATCH_SHAS : [1]str = ["11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267"]; -pub const RUN_ARTIFACT_URLS : [1]str = [""]; -pub const RUN_NOTES : [1]str = ["The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed."]; +pub const RUN_COUNT : u8 = 7; +pub const RUN_IDS : [7]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z"]; +pub const RUN_EXPERIMENT_IDS : [7]str = ["t27-4328-validate-trits", "t27-4614-count-buttons", "t27-4614-count-buttons", "t27-4695-tick", "t27-4695-tick", "t27-4613-validate-trits", "t27-4613-validate-trits"]; +pub const RUN_CONFIG_IDS : [7]str = ["bee-baseline", "bee-baseline", "bee-tri", "bee-baseline", "bee-tri", "bee-baseline", "bee-tri"]; +pub const RUN_STARTED_AT : [7]str = ["2026-09-23T04:10:25Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z"]; +pub const RUN_FINISHED_AT : [7]str = ["2026-09-23T04:32:27Z", "2026-09-26T18:54:00Z", "2026-09-26T18:54:02Z", "2026-09-26T18:55:26Z", "2026-09-26T18:55:26Z", "2026-09-26T18:57:10Z", "2026-09-26T18:56:37Z"]; +pub const RUN_STATES_BY_ID : [7]str = ["blocked", "passed", "passed", "passed", "passed", "passed", "passed"]; +pub const RUN_EVIDENCE : [7]str = ["SESSION-OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const RUN_VERDICTS : [7]str = ["inconclusive", "accepted", "accepted", "accepted", "accepted", "accepted", "accepted"]; +pub const RUN_LOG_SHAS : [7]str = ["", "6561a0821694e2a4afa55ddcfe9c4d960223c77487737f0d7e636c713e5f3950", "a4db3eb859fb458f0ba512b119ac41c4e350d2d111597927d5f032b80e9ca096", "444d27dc87851aaaf5ba7a642df4f6838bbcba043ab99c08dfc47baea5318f24", "4a8ea205b62bede5c36157fe7dc3c7536966c0f6580398f1ce104aef3dd55dc3", "25689dbaf0221824f23f6bcb7ec0f8556fec2cf40832ab226c68333d817b3805", "48716f028e82cc6bc95fe51e3c46de1803a47f2480292ab76c9356d86337dd7a"]; +pub const RUN_PATCH_SHAS : [7]str = ["11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267", "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", "cf7115ad1f649cc302646c3713488f21d810c47c3f01e1d01a7d72021a889f6b", "aa0443796cf2e7a0ca10dd1b5f885883e6ecaff97d5be77d50a994983e81b7ce", "6f44011b4ff970ffd2536c18baf755b64ea69f01539bde11243bcc1fcc743e96", "6c3e1562d398c7e59aacb44b09216463456f295354163590154d42279ded636a"]; +pub const RUN_ARTIFACT_URLS : [7]str = ["", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z"]; +pub const RUN_NOTES : [7]str = ["The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed.", "Added test count_buttons_is_4 without a compiler; every criterion passed when judged.", "Same patch as the baseline, byte for byte; also ran a mutation check of its own before reporting.", "Added test_tick_advances_one_cycle after reset(), argued from the compiler source; every criterion passed when judged.", "Added test_tick relative to the entry state; found that assigning a module var in a test body generates a shadowing local.", "Added validate_trits_check with five asserts; the four criteria pass, and test-report stays BLOCKED as at the base.", "Added validate_trits_check with six asserts; traced the pre-existing BLOCKED to mismatched trit encoders and decoders."]; // EAV measurements keep unknowns absent instead of turning them into misleading zeroes. -pub const MEASUREMENT_COUNT : u8 = 5; -pub const MEASUREMENT_RUN_IDS : [5]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z"]; -pub const MEASUREMENT_KEYS : [5]str = ["acceptance", "queen-verdict", "elapsed", "retries", "patch-lines"]; -pub const MEASUREMENT_VALUES : [5]str = ["BLOCKED", "inconclusive", "1322000", "2", "9"]; -pub const MEASUREMENT_UNITS : [5]str = ["verdict", "verdict", "ms", "count", "lines"]; -pub const MEASUREMENT_EVIDENCE : [5]str = ["SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED"]; -pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed", "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed", "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z", "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed", "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS"]; +pub const MEASUREMENT_COUNT : u8 = 53; +pub const MEASUREMENT_RUN_IDS : [53]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z"]; +pub const MEASUREMENT_KEYS : [53]str = ["acceptance", "queen-verdict", "elapsed", "retries", "patch-lines", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed"]; +pub const MEASUREMENT_VALUES : [53]str = ["BLOCKED", "inconclusive", "1322000", "2", "9", "PASSED", "accepted", "62703", "10", "4", "76852", "0", "3", "PASSED", "accepted", "64182", "14", "4", "77744", "8", "3", "PASSED", "accepted", "148105", "30", "16", "100044", "0", "4", "PASSED", "accepted", "148367", "27", "16", "98507", "9", "4", "PASSED", "accepted", "252696", "38", "12", "128055", "0", "2", "PASSED", "accepted", "219864", "27", "18", "113494", "17", "2"]; +pub const MEASUREMENT_UNITS : [53]str = ["verdict", "verdict", "ms", "count", "lines", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count"]; +pub const MEASUREMENT_EVIDENCE : [53]str = ["SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const MEASUREMENT_SOURCES : [53]str = ["session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed", "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed", "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z", "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed", "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 12 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 18 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile"]; test protocol_has_one_variable { assert REAL_GITHUB_TASKS_ONLY == true; @@ -109,27 +113,46 @@ test protocol_has_one_variable { assert TRI_ROLE_EVIDENCE == "OBSERVED"; } -test arena_names_all_four_lanes { - assert CONFIG_COUNT == 4; +test arena_names_all_five_lanes { + assert CONFIG_COUNT == 5; assert CONFIG_IDS[0] == "bee-baseline"; - assert CONFIG_IDS[1] == "bee-jev"; - assert CONFIG_IDS[2] == "igla-coder"; - assert CONFIG_IDS[3] == "igla-race"; - assert CONFIG_EVIDENCE[1] == "SOURCE-CLAIM"; - assert CONFIG_STATE_EVIDENCE[1] == "OBSERVED"; + assert CONFIG_IDS[1] == "bee-tri"; + assert CONFIG_IDS[2] == "bee-jev"; + assert CONFIG_IDS[3] == "igla-coder"; + assert CONFIG_IDS[4] == "igla-race"; + assert CONFIG_EVIDENCE[1] == "OBSERVED"; + assert CONFIG_EVIDENCE[2] == "SOURCE-CLAIM"; + assert CONFIG_STATE_EVIDENCE[2] == "OBSERVED"; } test first_task_is_real_and_pinned { - assert EXPERIMENT_COUNT == 1; + assert EXPERIMENT_COUNT == 4; assert EXPERIMENT_ISSUE_NUMBERS[0] == "4328"; assert EXPERIMENT_REPOS[0] == "gHashTag/t27"; assert EXPERIMENT_STATES_BY_ID[0] == "credential-blocked"; assert EXPERIMENT_MODEL_EVIDENCE[0] == "UNKNOWN"; } +test campaign_20260926_is_paired_and_unranked { + assert EXPERIMENT_ISSUE_NUMBERS[1] == "4614"; + assert EXPERIMENT_ISSUE_NUMBERS[2] == "4695"; + assert EXPERIMENT_ISSUE_NUMBERS[3] == "4613"; + assert EXPERIMENT_STATES_BY_ID[3] == "judged"; + assert EXPERIMENT_MODEL_EVIDENCE[3] == "SESSION-OBSERVED"; + assert RUN_COUNT == 7; + assert RUN_CONFIG_IDS[1] == "bee-baseline"; + assert RUN_CONFIG_IDS[2] == "bee-tri"; + assert RUN_EXPERIMENT_IDS[1] == RUN_EXPERIMENT_IDS[2]; + assert RUN_PATCH_SHAS[1] == RUN_PATCH_SHAS[2]; + assert MEASUREMENT_COUNT == 53; +} + test metric_vocabulary_has_jev_and_cost_evidence { - assert METRIC_COUNT == 11; + assert METRIC_COUNT == 14; assert METRIC_KEYS[8] == "cost"; assert METRIC_KEYS[9] == "jev-latency"; assert METRIC_KEYS[10] == "jev-confidence"; + assert METRIC_KEYS[11] == "total-tokens"; + assert METRIC_KEYS[12] == "judge-calls"; + assert METRIC_KEYS[13] == "mutants-killed"; } diff --git a/apps/website/public/roadmap/goals.json b/apps/website/public/roadmap/goals.json index f2cf84b542..46c783028d 100644 --- a/apps/website/public/roadmap/goals.json +++ b/apps/website/public/roadmap/goals.json @@ -65,7 +65,11 @@ "Rust" ], "target": "t27c gen-rust (servers); UI stays for stage 8", - "issue": 4545 + "issue": 4545, + "locked": { + "en": "its repository is private, and the roadmap feeder cannot read it until the ROADMAP_READ_TOKEN secret is set", + "ru": "репозиторий закрытый, и фидер роадмапа не может его читать, пока не задан секрет ROADMAP_READ_TOKEN" + } }, { "id": "vibee", @@ -88,7 +92,11 @@ "Erlang" ], "target": "t27c gen-rust", - "issue": 4546 + "issue": 4546, + "locked": { + "en": "its repository is private, and the roadmap feeder cannot read it until the ROADMAP_READ_TOKEN secret is set", + "ru": "репозиторий закрытый, и фидер роадмапа не может его читать, пока не задан секрет ROADMAP_READ_TOKEN" + } }, { "id": "core", @@ -178,7 +186,45 @@ "Swift" ], "target": "undecided: t27c has no interface target", - "issue": 4550 + "issue": 4550, + "locked": { + "en": "it needs a decision first: t27c has no interface target", + "ru": "сначала нужно решение: у t27c нет интерфейсной цели" + } + }, + { + "id": "endgame", + "stage": 9, + "title": { + "en": "Endgame: the browser and every dependency", + "ru": "Эндшпиль: браузер и каждая зависимость" + }, + "why": { + "en": "After stage 8 the comb keeps widening to its top row: all of BrowserOS - the browser itself, not only trios/agent-server - and every third-party library the stack runs in another language, rewritten in .t27. The browser is Chromium 146.0.7680.31, which BrowserOS patches and fetches at build time. Measured 2026-09-27 by the same rules as the count above, Chromium's src is 1,145 MB, V8 is 146 MB and Skia is 51 MB. That is seven times everything the count above holds. Blink (132 MB outside its tests) and 258 more DEPS repositories are not in that figure. The running services also lock 2,187 npm packages, 650 crates and 35 Go modules.", + "ru": "После этапа 8 соты расширяются до верхнего ряда: весь BrowserOS - сам браузер, а не только trios/agent-server - и каждая сторонняя библиотека на другом языке, на которой работает стек, переписанные на .t27. Браузер - это Chromium 146.0.7680.31: BrowserOS патчит его и скачивает при сборке. Измерено 2026-09-27 по тем же правилам, что и подсчёт выше: src Chromium - 1145 МБ, V8 - 146 МБ, Skia - 51 МБ. Это в семь раз больше всего, что входит в подсчёт выше. Blink (132 МБ без тестов) и ещё 258 репозиториев из DEPS в эту цифру не входят. Кроме того, работающие сервисы фиксируют в lock-файлах 2187 npm-пакетов, 650 crates и 35 модулей Go." + }, + "repos": [ + "gHashTag/BrowserOS" + ], + "languages": [ + "C++", + "C", + "Java", + "Objective-C", + "TypeScript", + "JavaScript" + ], + "target": "t27c gen-rust, gen-c", + "issue": 4858, + "locked": { + "en": "after stage 8; t27c has no C++ target", + "ru": "после этапа 8; у t27c нет цели C++" + }, + "measured": { + "bytes": 1341665569, + "at": "2026-09-27", + "source": "Chromium 146.0.7680.31 src @4d32251, V8 14.6.202.6 @0a35ee1 and Skia @9022820, the commits Chromium pins; read from the github.com/chromium/chromium, v8/v8 and google/skia mirrors and counted by the roadmap-stack.mjs rules, without Config, Docker and Make" + } } ] -} \ No newline at end of file +} diff --git a/apps/website/qa/queen-contrast-contract.mjs b/apps/website/qa/queen-contrast-contract.mjs index 28936fee9b..29cc498943 100644 --- a/apps/website/qa/queen-contrast-contract.mjs +++ b/apps/website/qa/queen-contrast-contract.mjs @@ -221,6 +221,10 @@ const REACHED = { /* PEOPLE, the contributors half of that tab: the fourth sheet the list has caught on its first run, which is four for four. */ 'src/components/QueenPeople.css', + /* LEVEL II, the comb on the ROADMAP view: five for five. It renders inside + .rm, whose opaque gradient already grounds it, and its own panel is + opaque on top of that. */ + 'src/components/queenRoadmapGame.css', ], 'src/App.tsx': [ 'src/components/AgiGameBlock.css', @@ -833,6 +837,8 @@ const TOKEN_SCOPES = new Map([ ['.queen27-page.is-shell:has(.queen-chat-tab)', 'board'], // Views and components mounted inside it. ['.rm', 'board'], + // LEVEL II, the comb: mounted inside `.rm`. + ['.rg', 'board'], ['.queen-catalog-layer', 'board'], ['.queen-catalog-layer:has(.queen-catalog-toolbar.is-search-open)', 'board'], ['.queen-hive-display', 'board'], @@ -1673,6 +1679,18 @@ const SEALED = { '.rm-hero': '.rm', '.rm-kpis div': '.rm', '.rm-stages li': '.rm', + + // LEVEL II, the comb, renders as the first child of `.rm` + // (QueenRoadmap.tsx), so its tiles sit on the same opaque wash as the three + // above; the hive stops at `.rm`'s boundary. + '.rg': '.rm', + '.rg-hud div': '.rm', + '.rg-targets li': '.rm', + '.rg-sector': '.rm', + // The raid banner and the boss cards: the same view root, the same wash. + '.rg-raid': '.rm', + '.rg-raid.is-none': '.rm', + '.rg-boss': '.rm', } /** Surfaces that buy legibility with a HALO instead of with a ground. diff --git a/apps/website/qa/queen-wars-contract.mjs b/apps/website/qa/queen-wars-contract.mjs index a679eb386c..07e0bec200 100644 --- a/apps/website/qa/queen-wars-contract.mjs +++ b/apps/website/qa/queen-wars-contract.mjs @@ -33,12 +33,16 @@ assert.match(spec, /pub const MEASUREMENT_COUNT : u8 = \d+;/) const generated = JSON.parse(read(JSON_OUT)) assert.equal(generated.source.spec, SPEC) assert.match(generated.source.sha256, /^[0-9a-f]{64}$/) -assert.deepEqual(generated.configurations.map((item) => item.id), ['bee-baseline', 'bee-jev', 'igla-coder', 'igla-race']) +assert.deepEqual(generated.configurations.map((item) => item.id), ['bee-baseline', 'bee-tri', 'bee-jev', 'igla-coder', 'igla-race']) assert.equal(generated.protocol.variableFactor, 'tri-decision-layer') assert.equal(generated.protocol.triRoleEvidence, 'OBSERVED') assert.match(generated.protocol.triRoleSource, /^https:\/\/t27\.ai\//) -assert.equal(generated.configurations[1].evidence, 'SOURCE-CLAIM') -assert.equal(generated.configurations[1].stateEvidence, 'OBSERVED') +const jev = generated.configurations.find((item) => item.id === 'bee-jev') +assert.equal(jev.evidence, 'SOURCE-CLAIM') +assert.equal(jev.stateEvidence, 'OBSERVED') +for (const run of generated.runs.filter((item) => item.evidence === 'OBSERVED')) { + assert.match(run.artifactUrl, /^https:\/\/github\.com\/gHashTag\/trinity\/tree\/[0-9a-f]{40}\//, `${run.id} is observed but its artifacts are not pinned to a commit`) +} assert.equal(generated.experiments[0].issue.url, 'https://github.com/gHashTag/t27/issues/4328') assert.equal(generated.experiments[0].baseSha.length, 40) assert.equal(generated.experiments[0].modelEvidence, 'UNKNOWN') diff --git a/apps/website/qa/roadmap-game-contract.mjs b/apps/website/qa/roadmap-game-contract.mjs new file mode 100644 index 0000000000..2219acb5af --- /dev/null +++ b/apps/website/qa/roadmap-game-contract.mjs @@ -0,0 +1,107 @@ +// The rules of LEVEL II (src/lib/roadmapGame.ts), held by example. +// +// The one rule that crosses a repository boundary is the raid: this page and +// gHashTag/t27 tools/queen/feed_roadmap.py must name the same sector on the +// same UTC day, or the board says "raid" over a sector the feeder is not +// feeding first. The feeder pins RAID_STAGES = (1, 2, 5, 6, 7) in its own +// self-test; this contract pins the same list out of goals.json. + +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { + BOSS_OPENS_AT, + builtShareBelow, + crackedTargets, + honeyOf, + msToNextDay, + parsePortTitle, + pulsePhase, + raidOf, + raidStages, + rowsFor, + targetOf, + utcDay, +} from '../src/lib/roadmapGame.ts' + +const goals = JSON.parse(readFileSync(new URL('../public/roadmap/goals.json', import.meta.url), 'utf8')).goals + +// The raid: the same list and the same formula as the feeder. +assert.deepEqual(raidStages(goals), [1, 2, 5, 6, 7], 'raid stages drifted from feed_roadmap.py RAID_STAGES') +const day = 20358 // 2025-09-27 UTC +assert.equal(utcDay(Date.UTC(2025, 8, 27, 23, 59)), day) +assert.equal(raidOf(goals, day), [1, 2, 5, 6, 7][day % 5]) +assert.deepEqual( + [0, 1, 2, 3, 4].map((k) => raidOf(goals, day + k)).sort(), + [1, 2, 5, 6, 7], + 'five days visit every raidable stage once', +) +assert.equal(raidOf([{ stage: 8 }, { stage: 3, locked: { en: 'x', ru: 'x' } }], day), null) +assert.equal(msToNextDay(Date.UTC(2025, 8, 27, 23, 0)), 3_600_000) + +// Honey: functions of built and cracked cells; raid cells closed today count twice. +const now = Date.UTC(2025, 8, 27, 12) +const today = new Date(now).toISOString() +const yesterday = new Date(now - 86_400_000).toISOString() +const tasks = [ + { stage: 1, units: 2, state: 'built', closedAt: today }, + { stage: 1, units: 3, state: 'built', closedAt: yesterday }, + { stage: 2, units: 5, state: 'cracked', closedAt: today }, + { stage: 2, units: 7, state: 'target' }, + { stage: 5, units: 1, state: 'building' }, +] +assert.equal(honeyOf(tasks, 1, utcDay(now)), 2 * 2 + 3 + 5) +assert.equal(honeyOf(tasks, null, utcDay(now)), 2 + 3 + 5) + +// Bosses open on half of the comb below them. +assert.equal(BOSS_OPENS_AT, 0.5) +assert.equal(builtShareBelow(tasks), 3 / 5) +assert.equal(builtShareBelow([{ stage: 9, units: 1, state: 'built' }]), 0, 'the endgame is not below the bosses') +assert.equal(builtShareBelow([]), 0) + +// A port task's target, from both title shapes the feeders write. +assert.equal( + targetOf('Port tools/check_x.py (Python, 4 functions) to specs/port/tools/check_x.t27'), + 'specs/port/tools/check_x.t27', +) +assert.equal( + targetOf('Port part 1 of 2 of tools/y.py to specs/port/tools/y.t27 (8 functions)'), + 'specs/port/tools/y.t27', +) +assert.equal(targetOf('Implement the 8 empty function bodies in specs/vsa/packed_vsa.t27'), null) + +// Cracks: defects that name a port file; a port task naming its own target is not one. +const cracks = crackedTargets([ + 'Port tools/a.py (Python, 1 function) to specs/port/tools/a.t27', + 'Implement the 1 empty function body in specs/port/tools/b.t27', + 'specs/port/trinity/src/vm/opcodes.t27: test-report BLOCKED on master', +]) +assert.deepEqual([...cracks].sort(), ['specs/port/tools/b.t27', 'specs/port/trinity/src/vm/opcodes.t27']) + +// The round's pulse, from the board's own numbers. +assert.equal(pulsePhase(new Date(now - 30_000).toISOString(), 60, now), 0.5) +assert.equal(pulsePhase(new Date(now - 90_000).toISOString(), 60, now), 0.5, 'a missed round wraps, it does not overflow') +assert.equal(pulsePhase(null, 60, now), null) +assert.equal(pulsePhase(today, 0, now), null) +assert.equal(pulsePhase('not a time', 60, now), null) + +// Titles: the stage-1 shape, the cross-repository shape, and a count-less title. +assert.deepEqual(parsePortTitle('Port tools/toolbelt.py (Python, 5 functions) to specs/port/tools/toolbelt.t27'), { + repo: 'gHashTag/t27', path: 'tools/toolbelt.py', lang: 'Python', units: 5, +}) +assert.deepEqual( + parsePortTitle('Port gHashTag/BrowserOS:trios/a/queen-lease.ts (TypeScript, 6 functions) to specs/port/browseros/trios/a/queen-lease.t27'), + { repo: 'gHashTag/BrowserOS', path: 'trios/a/queen-lease.ts', lang: 'TypeScript', units: 6 }, +) +assert.deepEqual(parsePortTitle('Port fpga/vivado/blinky.v (Verilog, 1 module) to specs/port/fpga/vivado/blinky.t27'), { + repo: 'gHashTag/t27', path: 'fpga/vivado/blinky.v', lang: 'Verilog', units: 1, +}) +assert.equal(parsePortTitle('Port part 1 of 2 of tools/y.py to specs/port/tools/y.t27').units, 4, 'no count reads as middling') +assert.equal(parsePortTitle('Implement something'), null) + +// The comb: apex-down rows hold 1, 2, 3 ... cells; 9 rows at least, 16 at most. +assert.equal(rowsFor(1), 9) +assert.equal(rowsFor(45), 9) +assert.equal(rowsFor(46), 10) +assert.equal(rowsFor(10_000), 16) + +console.log('roadmap-game contract: raid, honey, bosses, cracks, pulse, titles and rows hold') diff --git a/apps/website/scripts/queen-wars-from-spec.mjs b/apps/website/scripts/queen-wars-from-spec.mjs index c35d0ddfea..cbb315e356 100644 --- a/apps/website/scripts/queen-wars-from-spec.mjs +++ b/apps/website/scripts/queen-wars-from-spec.mjs @@ -50,10 +50,10 @@ const EXPECTED = { GENERATED: [TS_OUT, JSON_OUT, PUBLIC_SPEC_OUT], EVIDENCE_LEVELS: ['OBSERVED', 'SESSION-OBSERVED', 'SOURCE-CLAIM', 'TARGET', 'UNKNOWN'], CONFIG_STATES: ['ready', 'credential-blocked', 'checkpoint-unverified', 'pipeline-only'], - EXPERIMENT_STATES: ['planned', 'running', 'credential-blocked', 'complete', 'invalid'], + EXPERIMENT_STATES: ['planned', 'running', 'credential-blocked', 'judged', 'complete', 'invalid'], RUN_STATES: ['pending', 'running', 'passed', 'failed', 'blocked'], VERDICTS: ['accepted', 'rejected', 'inconclusive', 'not-reviewed'], - CONFIG_IDS: ['bee-baseline', 'bee-jev', 'igla-coder', 'igla-race'], + CONFIG_IDS: ['bee-baseline', 'bee-tri', 'bee-jev', 'igla-coder', 'igla-race'], } const PARALLEL = { @@ -77,6 +77,9 @@ const ACCEPTANCE_BY_RUN_STATE = new Map([ ['blocked', 'BLOCKED'], ]) const COMPARABLE_RUN_STATES = new Set(['passed', 'failed']) +// The variable factor is the TRI decision layer, so a judged or complete +// experiment needs a sealed run of both sides of it. JEV is a comparison arm only. +const PAIRED_ARMS = ['bee-baseline', 'bee-tri'] const REVIEWED_VERDICTS = new Set(['accepted', 'rejected']) const NONNEGATIVE_NUMBER_UNITS = new Set(['ms', 'usd']) const NONNEGATIVE_INTEGER_UNITS = new Set(['tokens', 'count', 'lines']) @@ -193,12 +196,13 @@ export function semanticProblems(f, file = WARS_SPEC) { } for (let i = 0; i < f.EXPERIMENT_COUNT; i++) { - if (f.EXPERIMENT_STATES_BY_ID[i] !== 'complete') continue + const experimentState = f.EXPERIMENT_STATES_BY_ID[i] + if (experimentState !== 'complete' && experimentState !== 'judged') continue const experimentId = f.EXPERIMENT_IDS[i] - if (f.EXPERIMENT_MODEL_EVIDENCE[i] !== 'OBSERVED' || /\bUNKNOWN\b/i.test(f.EXPERIMENT_EXECUTOR_MODELS[i])) { + if (experimentState === 'complete' && (f.EXPERIMENT_MODEL_EVIDENCE[i] !== 'OBSERVED' || /\bUNKNOWN\b/i.test(f.EXPERIMENT_EXECUTOR_MODELS[i]))) { p.push(`${file}: complete experiment ${experimentId} requires an explicit observed executor model and reasoning configuration`) } - for (const required of ['bee-baseline', 'bee-jev']) { + for (const required of PAIRED_ARMS) { const sealedRun = f.RUN_EXPERIMENT_IDS.some((id, at) => { if (id !== experimentId || f.RUN_CONFIG_IDS[at] !== required) return false const runId = f.RUN_IDS[at] @@ -218,7 +222,7 @@ export function semanticProblems(f, file = WARS_SPEC) { && f.MEASUREMENT_EVIDENCE[verdictAt] === 'OBSERVED' }) if (!sealedRun) { - p.push(`${file}: complete experiment ${experimentId} has no sealed ${required} run (passed or failed state, accepted or rejected Queen verdict, observed evidence, timestamps, log SHA, patch SHA, https artifact, acceptance and queen-verdict measurements required)`) + p.push(`${file}: ${experimentState} experiment ${experimentId} has no sealed ${required} run (passed or failed state, accepted or rejected Queen verdict, observed evidence, timestamps, log SHA, patch SHA, https artifact, acceptance and queen-verdict measurements required)`) } } } diff --git a/apps/website/scripts/queen-wars-from-spec.test.mjs b/apps/website/scripts/queen-wars-from-spec.test.mjs index 3580916729..9c61fa9428 100644 --- a/apps/website/scripts/queen-wars-from-spec.test.mjs +++ b/apps/website/scripts/queen-wars-from-spec.test.mjs @@ -54,15 +54,21 @@ test('the WARS source compiles, evaluates its tests and renders all projections' assert.ok(out.tests.asserts >= 15) assert.equal(out.specSha, sha256(Buffer.from(source, 'utf8'))) assert.equal(out.arena.source.sha256, out.specSha) - assert.equal(out.arena.configurations.length, 4) - assert.equal(out.arena.configurations[1].evidence, 'SOURCE-CLAIM') - assert.equal(out.arena.configurations[1].stateEvidence, 'OBSERVED') - assert.match(out.arena.configurations[1].source, /^https:\/\/docs\.typesafe\.ai\//) + assert.equal(out.arena.configurations.length, 5) + const jev = out.arena.configurations.find((config) => config.id === 'bee-jev') + assert.equal(jev.evidence, 'SOURCE-CLAIM') + assert.equal(jev.stateEvidence, 'OBSERVED') + assert.match(jev.source, /^https:\/\/docs\.typesafe\.ai\//) + assert.equal(out.arena.configurations[1].id, 'bee-tri') assert.equal(out.arena.experiments[0].issue.number, 4328) assert.equal(out.arena.experiments[0].modelEvidence, 'UNKNOWN') assert.equal(out.arena.protocol.triRoleEvidence, 'OBSERVED') - assert.equal(out.arena.runs.length, 1) - assert.equal(out.arena.measurements.length, 5) + // The 2026-09-26 campaign: three judged experiments, each with a sealed + // baseline and TRI run, and no experiment claims a winner. + assert.deepEqual(out.arena.experiments.slice(1).map((e) => e.state), ['judged', 'judged', 'judged']) + assert.equal(out.arena.experiments.some((e) => e.state === 'complete'), false) + assert.equal(out.arena.runs.length, 7) + assert.equal(out.arena.measurements.length, 53) }) test('committed projections are deterministic and current', async () => { @@ -212,13 +218,13 @@ test('zero and probability boundaries are valid measured values', async () => { assert.deepEqual(semanticProblems(fields, 'numeric-boundaries.t27'), []) }) -test('a complete experiment requires sealed baseline and JEV runs with both decision measurements', async () => { +test('a complete experiment requires sealed baseline and TRI runs with both decision measurements', async () => { const fields = await parsedFields() fields.EXPERIMENT_STATES_BY_ID[0] = 'complete' appendRun(fields, { - id: 't27-4328-bee-jev-pending', + id: 't27-4328-bee-tri-pending', experimentId: fields.EXPERIMENT_IDS[0], - configId: 'bee-jev', + configId: 'bee-tri', startedAt: '', finishedAt: '', state: 'pending', @@ -232,7 +238,7 @@ test('a complete experiment requires sealed baseline and JEV runs with both deci const problems = semanticProblems(fields, 'complete.t27') assert.ok(problems.some((problem) => /complete experiment .* sealed bee-baseline run/.test(problem)), problems.join('\n')) - assert.ok(problems.some((problem) => /complete experiment .* sealed bee-jev run/.test(problem)), problems.join('\n')) + assert.ok(problems.some((problem) => /complete experiment .* sealed bee-tri run/.test(problem)), problems.join('\n')) }) test('blocked and inconclusive runs remain valid evidence while an experiment is not complete', async () => { @@ -245,11 +251,11 @@ test('blocked and inconclusive runs remain valid evidence while an experiment is test('blocked and inconclusive terminal runs cannot claim a complete comparison', async () => { const fields = await parsedFields() fields.EXPERIMENT_STATES_BY_ID[0] = 'complete' - const runId = 't27-4328-bee-jev-blocked' + const runId = 't27-4328-bee-tri-blocked' appendRun(fields, { id: runId, experimentId: fields.EXPERIMENT_IDS[0], - configId: 'bee-jev', + configId: 'bee-tri', startedAt: '2026-09-23T05:00:00Z', finishedAt: '2026-09-23T05:01:00Z', state: 'blocked', @@ -279,7 +285,7 @@ test('blocked and inconclusive terminal runs cannot claim a complete comparison' const problems = semanticProblems(fields, 'blocked-complete.t27') assert.ok(problems.some((problem) => /complete experiment .* sealed bee-baseline run/.test(problem)), problems.join('\n')) - assert.ok(problems.some((problem) => /complete experiment .* sealed bee-jev run/.test(problem)), problems.join('\n')) + assert.ok(problems.some((problem) => /complete experiment .* sealed bee-tri run/.test(problem)), problems.join('\n')) }) test('a comparison cannot complete with unknown model identity or unsealed arms', async () => { @@ -290,11 +296,11 @@ test('a comparison cannot complete with unknown model identity or unsealed arms' fields.MEASUREMENT_VALUES[0] = 'PASSED' fields.MEASUREMENT_VALUES[1] = 'accepted' - const runId = 't27-4328-bee-jev-failed' + const runId = 't27-4328-bee-tri-failed' appendRun(fields, { id: runId, experimentId: fields.EXPERIMENT_IDS[0], - configId: 'bee-jev', + configId: 'bee-tri', startedAt: '2026-09-23T05:00:00Z', finishedAt: '2026-09-23T05:01:00Z', state: 'failed', @@ -325,7 +331,7 @@ test('a comparison cannot complete with unknown model identity or unsealed arms' const problems = semanticProblems(fields, 'unsealed-complete.t27') assert.ok(problems.some((problem) => /explicit observed executor model/.test(problem)), problems.join('\n')) assert.ok(problems.some((problem) => /sealed bee-baseline run/.test(problem)), problems.join('\n')) - assert.ok(problems.some((problem) => /sealed bee-jev run/.test(problem)), problems.join('\n')) + assert.ok(problems.some((problem) => /sealed bee-tri run/.test(problem)), problems.join('\n')) }) test('explicit model identity and sealed observed arms can complete a comparison', async () => { @@ -343,11 +349,11 @@ test('explicit model identity and sealed observed arms can complete a comparison fields.MEASUREMENT_EVIDENCE[0] = 'OBSERVED' fields.MEASUREMENT_EVIDENCE[1] = 'OBSERVED' - const runId = 't27-4328-bee-jev-failed-sealed' + const runId = 't27-4328-bee-tri-failed-sealed' appendRun(fields, { id: runId, experimentId: fields.EXPERIMENT_IDS[0], - configId: 'bee-jev', + configId: 'bee-tri', startedAt: '2026-09-23T05:00:00Z', finishedAt: '2026-09-23T05:01:00Z', state: 'failed', @@ -384,3 +390,20 @@ test('non-ASCII source is refused by the L3 gate', async () => { assert.equal(out.ts, null) assert.ok(out.problems.some((problem) => /non-ASCII/.test(problem)), out.problems.join('\n')) }) + +test('a judged experiment needs a sealed TRI run, and cannot be promoted to complete without a model id', async () => { + const fields = await parsedFields() + const experimentId = 't27-4613-validate-trits' + const triAt = fields.RUN_IDS.findIndex((id, at) => fields.RUN_EXPERIMENT_IDS[at] === experimentId && fields.RUN_CONFIG_IDS[at] === 'bee-tri') + assert.ok(triAt > 0) + + const unsealed = structuredClone(fields) + unsealed.RUN_ARTIFACT_URLS[triAt] = '' + let problems = semanticProblems(unsealed, 'judged.t27') + assert.ok(problems.some((problem) => /judged experiment t27-4613-validate-trits has no sealed bee-tri run/.test(problem)), problems.join('\n')) + + const promoted = structuredClone(fields) + promoted.EXPERIMENT_STATES_BY_ID[fields.EXPERIMENT_IDS.indexOf(experimentId)] = 'complete' + problems = semanticProblems(promoted, 'promoted.t27') + assert.ok(problems.some((problem) => /complete experiment t27-4613-validate-trits requires an explicit observed executor model/.test(problem)), problems.join('\n')) +}) diff --git a/apps/website/specs/queen/wars.t27 b/apps/website/specs/queen/wars.t27 index a49983c9fb..3fa6db22f3 100644 --- a/apps/website/specs/queen/wars.t27 +++ b/apps/website/specs/queen/wars.t27 @@ -16,9 +16,11 @@ pub const SCHEMA_VERSION : u8 = 2; pub const GENERATED : [3]str = ["src/lib/queenWars.generated.ts", "public/queen/wars.json", "public/queen/wars.t27"]; // Epistemic and lifecycle vocabularies. Every displayed claim uses one of these. +// An experiment is "judged" when every arm ran and the acceptance gates judged it, +// but no winner may be declared (for example, the executor model id is not recorded). pub const EVIDENCE_LEVELS : [5]str = ["OBSERVED", "SESSION-OBSERVED", "SOURCE-CLAIM", "TARGET", "UNKNOWN"]; pub const CONFIG_STATES : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; -pub const EXPERIMENT_STATES : [5]str = ["planned", "running", "credential-blocked", "complete", "invalid"]; +pub const EXPERIMENT_STATES : [6]str = ["planned", "running", "credential-blocked", "judged", "complete", "invalid"]; pub const RUN_STATES : [5]str = ["pending", "running", "passed", "failed", "blocked"]; pub const VERDICTS : [4]str = ["accepted", "rejected", "inconclusive", "not-reviewed"]; pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; @@ -38,67 +40,69 @@ pub const TRI_ROLE_SOURCE : str = "https://t27.ai/t27/files/trinity/apps/website // Measurement vocabulary. Values are strings so an exact native unit is preserved; // absent rows mean unknown. A numeric zero is a measured zero only when a source exists. -pub const METRIC_COUNT : u8 = 11; -pub const METRIC_KEYS : [11]str = ["acceptance", "queen-verdict", "elapsed", "input-tokens", "output-tokens", "tool-calls", "retries", "patch-lines", "cost", "jev-latency", "jev-confidence"]; -pub const METRIC_UNITS : [11]str = ["verdict", "verdict", "ms", "tokens", "tokens", "count", "count", "lines", "usd", "ms", "probability"]; +pub const METRIC_COUNT : u8 = 14; +pub const METRIC_KEYS : [14]str = ["acceptance", "queen-verdict", "elapsed", "input-tokens", "output-tokens", "tool-calls", "retries", "patch-lines", "cost", "jev-latency", "jev-confidence", "total-tokens", "judge-calls", "mutants-killed"]; +pub const METRIC_UNITS : [14]str = ["verdict", "verdict", "ms", "tokens", "tokens", "count", "count", "lines", "usd", "ms", "probability", "tokens", "count", "count"]; -// Arena configurations. IGLA entries are visible now, but neither is presented as a -// measured coding model until an executable checkpoint and a witnessed run exist. -pub const CONFIG_COUNT : u8 = 4; -pub const CONFIG_IDS : [4]str = ["bee-baseline", "bee-jev", "igla-coder", "igla-race"]; -pub const CONFIG_NAMES : [4]str = ["Bee baseline", "Bee + JEV", "IGLA CODER", "IGLA RACE"]; -pub const CONFIG_KINDS : [4]str = ["coding-agent", "coding-agent-plus-decision-layer", "model-training-target", "training-race-pipeline"]; -pub const CONFIG_STATES_BY_ID : [4]str = ["ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; -pub const CONFIG_EVIDENCE : [4]str = ["OBSERVED", "SOURCE-CLAIM", "OBSERVED", "OBSERVED"]; -pub const CONFIG_SOURCES : [4]str = ["current Codex task runtime", "https://docs.typesafe.ai/introduction/coding-agents", "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "https://github.com/gHashTag/trios-trainer-igla"]; -pub const CONFIG_NOTES : [4]str = ["Control arm: the coding Bee chooses and implements without an external decision model.", "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result."]; -pub const CONFIG_STATE_EVIDENCE : [4]str = ["OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; -pub const CONFIG_STATE_SOURCES : [4]str = ["current Codex task runtime on 2026-09-23", "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", "repository inspection of the linked IGLA RACE pipeline on 2026-09-23"]; -pub const CONFIG_STATE_NOTES : [4]str = ["The baseline Bee can run locally.", "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed.", "No executable checkpoint was verified for this arena.", "The training pipeline is visible, but no coding-model run is claimed."]; +// Arena configurations. bee-baseline and bee-tri are the paired arms: the only +// difference is the TRI decision layer (t27c available while the Bee works). JEV +// stays as a comparison arm. IGLA entries are visible now, but neither is presented +// as a measured coding model until an executable checkpoint and a witnessed run exist. +pub const CONFIG_COUNT : u8 = 5; +pub const CONFIG_IDS : [5]str = ["bee-baseline", "bee-tri", "bee-jev", "igla-coder", "igla-race"]; +pub const CONFIG_NAMES : [5]str = ["Bee baseline", "Bee + TRI", "Bee + JEV", "IGLA CODER", "IGLA RACE"]; +pub const CONFIG_KINDS : [5]str = ["coding-agent", "coding-agent-plus-compiler-judge", "coding-agent-plus-decision-layer", "model-training-target", "training-race-pipeline"]; +pub const CONFIG_STATES_BY_ID : [5]str = ["ready", "ready", "credential-blocked", "checkpoint-unverified", "pipeline-only"]; +pub const CONFIG_EVIDENCE : [5]str = ["OBSERVED", "OBSERVED", "SOURCE-CLAIM", "OBSERVED", "OBSERVED"]; +pub const CONFIG_SOURCES : [5]str = ["coding agent runtime: Codex task runtime (2026-09-23 run); Claude Code subagents (2026-09-26 campaign)", "t27c 0.4.0 built from the gHashTag/t27 bootstrap at afe2186cfa9572c8c2d519506aa85a8c9447eae5", "https://docs.typesafe.ai/introduction/coding-agents", "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "https://github.com/gHashTag/trios-trainer-igla"]; +pub const CONFIG_NOTES : [5]str = ["Control arm: the coding Bee chooses and implements without an external decision layer. In the 2026-09-26 campaign it had no t27c while it worked; the gates judged it afterwards.", "Variable arm: the same Bee, prompt, tools and budget, plus the t27c compiler as its decision layer. It may run read-only t27c subcommands to choose between candidate edits; t27c never writes the patch.", "Comparison arm only: the same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", "Observed training and evaluation repository. It is a pipeline competitor, not a runnable coding-model result."]; +pub const CONFIG_STATE_EVIDENCE : [5]str = ["OBSERVED", "OBSERVED", "OBSERVED", "SOURCE-CLAIM", "OBSERVED"]; +pub const CONFIG_STATE_SOURCES : [5]str = ["current Codex task runtime on 2026-09-23; Claude Code subagents on 2026-09-26", "Queen WARS campaign 2026-09-26: six paired arms on gHashTag/t27 issues 4614, 4695 and 4613", "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "gHashTag/igla-coder-gpu LEDGER.md and c_infer/model.bin.json, read 2026-09-26; the repository is private", "repository inspection of the linked IGLA RACE pipeline on 2026-09-23"]; +pub const CONFIG_STATE_NOTES : [5]str = ["The baseline Bee can run locally.", "Runs wherever t27c and zig 0.16.0 build; no credential is needed.", "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed.", "The IGLA pilot reports trained checkpoints (tern_tc: 9.08M parameters, 0.7613 bits per byte after 2.0B tokens; a 100M full-precision and ternary pair) and 2.0% and 1.3% pass@10 on a t27 fill-in-the-body bench. The checkpoints are not public, and none has run an arena issue.", "The training pipeline is visible, but no coding-model run is claimed."]; // Experiments. The exact prompt, tools, budget and acceptance live here, not in React. -pub const EXPERIMENT_COUNT : u8 = 1; -pub const EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; -pub const EXPERIMENT_NAMES : [1]str = ["validate_trits missing test"]; -pub const EXPERIMENT_REPOS : [1]str = ["gHashTag/t27"]; -pub const EXPERIMENT_ISSUE_URLS : [1]str = ["https://github.com/gHashTag/t27/issues/4328"]; -pub const EXPERIMENT_ISSUE_NUMBERS : [1]str = ["4328"]; -pub const EXPERIMENT_ISSUE_UPDATED_AT : [1]str = ["2026-09-20T13:51:01Z"]; -pub const EXPERIMENT_BASE_SHAS : [1]str = ["f123674fe40d6ba600a8c5c4948683198f756fdc"]; -pub const EXPERIMENT_EXECUTOR_MODELS : [1]str = ["UNKNOWN: exact model id and reasoning configuration were not exposed"]; -pub const EXPERIMENT_MODEL_EVIDENCE : [1]str = ["UNKNOWN"]; -pub const EXPERIMENT_MODEL_SOURCES : [1]str = ["current Codex task runtime did not expose a stable model id and reasoning configuration"]; -pub const EXPERIMENT_PROMPTS : [1]str = ["Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push."]; -pub const EXPERIMENT_TOOL_POLICIES : [1]str = ["Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push."]; -pub const EXPERIMENT_BUDGETS : [1]str = ["one Codex agent turn; no explicit token cap; record usage only when exposed"]; -pub const EXPERIMENT_ACCEPTANCE : [1]str = ["t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass"]; -pub const EXPERIMENT_STATES_BY_ID : [1]str = ["credential-blocked"]; -pub const EXPERIMENT_EVIDENCE : [1]str = ["OBSERVED"]; -pub const EXPERIMENT_NOTES : [1]str = ["This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed."]; +pub const EXPERIMENT_COUNT : u8 = 4; +pub const EXPERIMENT_IDS : [4]str = ["t27-4328-validate-trits", "t27-4614-count-buttons", "t27-4695-tick", "t27-4613-validate-trits"]; +pub const EXPERIMENT_NAMES : [4]str = ["validate_trits missing test", "count_buttons missing test", "tick missing test", "validate_trits missing test (re-filed 4328)"]; +pub const EXPERIMENT_REPOS : [4]str = ["gHashTag/t27", "gHashTag/t27", "gHashTag/t27", "gHashTag/t27"]; +pub const EXPERIMENT_ISSUE_URLS : [4]str = ["https://github.com/gHashTag/t27/issues/4328", "https://github.com/gHashTag/t27/issues/4614", "https://github.com/gHashTag/t27/issues/4695", "https://github.com/gHashTag/t27/issues/4613"]; +pub const EXPERIMENT_ISSUE_NUMBERS : [4]str = ["4328", "4614", "4695", "4613"]; +pub const EXPERIMENT_ISSUE_UPDATED_AT : [4]str = ["2026-09-20T13:51:01Z", "2026-09-23T09:31:53Z", "2026-09-24T02:32:52Z", "2026-09-23T09:31:51Z"]; +pub const EXPERIMENT_BASE_SHAS : [4]str = ["f123674fe40d6ba600a8c5c4948683198f756fdc", "afe2186cfa9572c8c2d519506aa85a8c9447eae5", "afe2186cfa9572c8c2d519506aa85a8c9447eae5", "afe2186cfa9572c8c2d519506aa85a8c9447eae5"]; +pub const EXPERIMENT_EXECUTOR_MODELS : [4]str = ["UNKNOWN: exact model id and reasoning configuration were not exposed", "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository"]; +pub const EXPERIMENT_MODEL_EVIDENCE : [4]str = ["UNKNOWN", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED"]; +pub const EXPERIMENT_MODEL_SOURCES : [4]str = ["current Codex task runtime did not expose a stable model id and reasoning configuration", "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it"]; +pub const EXPERIMENT_PROMPTS : [4]str = ["Resolve gHashTag/t27 issue 4328 at the pinned base. Inspect repository instructions. Add the smallest non-vacuous test coverage for validate_trits in specs/base/ternary_encoding.t27. Preserve every signature and function. Use TDD, run the issue acceptance commands and repository gates, and report exact evidence. Do not commit or push.", "Resolve gHashTag/t27 issue 4614 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4614.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", "Resolve gHashTag/t27 issue 4695 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4695.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", "Resolve gHashTag/t27 issue 4613 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4613.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED."]; +pub const EXPERIMENT_TOOL_POLICIES : [4]str = ["Local repository read, edit and test tools only; one isolated worktree; no network writes; no commit or push.", "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms."]; +pub const EXPERIMENT_BUDGETS : [4]str = ["one Codex agent turn; no explicit token cap; record usage only when exposed", "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime"]; +pub const EXPERIMENT_ACCEPTANCE : [4]str = ["t27c parse specs/base/ternary_encoding.t27; t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; t27c validate-vacuity and t27c test-report pass", "t27c coverage specs/boards/arty_a7.t27 reports Untested: 0; function count remains 5; test count is at least 18; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", "t27c coverage specs/fpga/testbench/simulator_tb.t27 reports Untested: 0; function count remains 4; test count is at least 7; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", "t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; the issue has no test-report criterion"]; +pub const EXPERIMENT_STATES_BY_ID : [4]str = ["credential-blocked", "judged", "judged", "judged"]; +pub const EXPERIMENT_EVIDENCE : [4]str = ["OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const EXPERIMENT_NOTES : [4]str = ["This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed.", "Both arms wrote the same 4-line test, byte for byte, and both kill 3 of 3 mutants of count_buttons. There is no difference to rank.", "Different tests; both are accepted and both kill 4 of 4 mutants of tick. The TRI arm also found a compiler defect: an assignment to a module var inside a test body is lowered as a new local, so the generated Zig fails with a shadowing error.", "Both arms pass the four criteria, but the spec is BLOCKED at the base and this issue has no test-report criterion, so the new test cannot run in the repository. In a review copy with the four pre-existing blockers removed, both tests pass and kill 2 of 4 mutants; the other 2 are caught at compile time by an existing invariant. The TRI arm traced the blocker to balanced-trit encoders paired with unipolar decoders."]; // Append-only run ledger. Completed, failed and blocked attempts all stay visible. -pub const RUN_COUNT : u8 = 1; -pub const RUN_IDS : [1]str = ["t27-4328-bee-baseline-20260923T041025Z"]; -pub const RUN_EXPERIMENT_IDS : [1]str = ["t27-4328-validate-trits"]; -pub const RUN_CONFIG_IDS : [1]str = ["bee-baseline"]; -pub const RUN_STARTED_AT : [1]str = ["2026-09-23T04:10:25Z"]; -pub const RUN_FINISHED_AT : [1]str = ["2026-09-23T04:32:27Z"]; -pub const RUN_STATES_BY_ID : [1]str = ["blocked"]; -pub const RUN_EVIDENCE : [1]str = ["SESSION-OBSERVED"]; -pub const RUN_VERDICTS : [1]str = ["inconclusive"]; -pub const RUN_LOG_SHAS : [1]str = [""]; -pub const RUN_PATCH_SHAS : [1]str = ["11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267"]; -pub const RUN_ARTIFACT_URLS : [1]str = [""]; -pub const RUN_NOTES : [1]str = ["The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed."]; +pub const RUN_COUNT : u8 = 7; +pub const RUN_IDS : [7]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z"]; +pub const RUN_EXPERIMENT_IDS : [7]str = ["t27-4328-validate-trits", "t27-4614-count-buttons", "t27-4614-count-buttons", "t27-4695-tick", "t27-4695-tick", "t27-4613-validate-trits", "t27-4613-validate-trits"]; +pub const RUN_CONFIG_IDS : [7]str = ["bee-baseline", "bee-baseline", "bee-tri", "bee-baseline", "bee-tri", "bee-baseline", "bee-tri"]; +pub const RUN_STARTED_AT : [7]str = ["2026-09-23T04:10:25Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z", "2026-09-26T18:52:57Z"]; +pub const RUN_FINISHED_AT : [7]str = ["2026-09-23T04:32:27Z", "2026-09-26T18:54:00Z", "2026-09-26T18:54:02Z", "2026-09-26T18:55:26Z", "2026-09-26T18:55:26Z", "2026-09-26T18:57:10Z", "2026-09-26T18:56:37Z"]; +pub const RUN_STATES_BY_ID : [7]str = ["blocked", "passed", "passed", "passed", "passed", "passed", "passed"]; +pub const RUN_EVIDENCE : [7]str = ["SESSION-OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const RUN_VERDICTS : [7]str = ["inconclusive", "accepted", "accepted", "accepted", "accepted", "accepted", "accepted"]; +pub const RUN_LOG_SHAS : [7]str = ["", "6561a0821694e2a4afa55ddcfe9c4d960223c77487737f0d7e636c713e5f3950", "a4db3eb859fb458f0ba512b119ac41c4e350d2d111597927d5f032b80e9ca096", "444d27dc87851aaaf5ba7a642df4f6838bbcba043ab99c08dfc47baea5318f24", "4a8ea205b62bede5c36157fe7dc3c7536966c0f6580398f1ce104aef3dd55dc3", "25689dbaf0221824f23f6bcb7ec0f8556fec2cf40832ab226c68333d817b3805", "48716f028e82cc6bc95fe51e3c46de1803a47f2480292ab76c9356d86337dd7a"]; +pub const RUN_PATCH_SHAS : [7]str = ["11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267", "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", "cf7115ad1f649cc302646c3713488f21d810c47c3f01e1d01a7d72021a889f6b", "aa0443796cf2e7a0ca10dd1b5f885883e6ecaff97d5be77d50a994983e81b7ce", "6f44011b4ff970ffd2536c18baf755b64ea69f01539bde11243bcc1fcc743e96", "6c3e1562d398c7e59aacb44b09216463456f295354163590154d42279ded636a"]; +pub const RUN_ARTIFACT_URLS : [7]str = ["", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z", "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z"]; +pub const RUN_NOTES : [7]str = ["The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed.", "Added test count_buttons_is_4 without a compiler; every criterion passed when judged.", "Same patch as the baseline, byte for byte; also ran a mutation check of its own before reporting.", "Added test_tick_advances_one_cycle after reset(), argued from the compiler source; every criterion passed when judged.", "Added test_tick relative to the entry state; found that assigning a module var in a test body generates a shadowing local.", "Added validate_trits_check with five asserts; the four criteria pass, and test-report stays BLOCKED as at the base.", "Added validate_trits_check with six asserts; traced the pre-existing BLOCKED to mismatched trit encoders and decoders."]; // EAV measurements keep unknowns absent instead of turning them into misleading zeroes. -pub const MEASUREMENT_COUNT : u8 = 5; -pub const MEASUREMENT_RUN_IDS : [5]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z"]; -pub const MEASUREMENT_KEYS : [5]str = ["acceptance", "queen-verdict", "elapsed", "retries", "patch-lines"]; -pub const MEASUREMENT_VALUES : [5]str = ["BLOCKED", "inconclusive", "1322000", "2", "9"]; -pub const MEASUREMENT_UNITS : [5]str = ["verdict", "verdict", "ms", "count", "lines"]; -pub const MEASUREMENT_EVIDENCE : [5]str = ["SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED"]; -pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed", "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed", "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z", "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed", "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS"]; +pub const MEASUREMENT_COUNT : u8 = 53; +pub const MEASUREMENT_RUN_IDS : [53]str = ["t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4328-bee-baseline-20260923T041025Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-baseline-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4614-bee-tri-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-baseline-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4695-bee-tri-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-baseline-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z", "t27-4613-bee-tri-20260926T185257Z"]; +pub const MEASUREMENT_KEYS : [53]str = ["acceptance", "queen-verdict", "elapsed", "retries", "patch-lines", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed", "acceptance", "queen-verdict", "elapsed", "tool-calls", "patch-lines", "total-tokens", "judge-calls", "mutants-killed"]; +pub const MEASUREMENT_VALUES : [53]str = ["BLOCKED", "inconclusive", "1322000", "2", "9", "PASSED", "accepted", "62703", "10", "4", "76852", "0", "3", "PASSED", "accepted", "64182", "14", "4", "77744", "8", "3", "PASSED", "accepted", "148105", "30", "16", "100044", "0", "4", "PASSED", "accepted", "148367", "27", "16", "98507", "9", "4", "PASSED", "accepted", "252696", "38", "12", "128055", "0", "2", "PASSED", "accepted", "219864", "27", "18", "113494", "17", "2"]; +pub const MEASUREMENT_UNITS : [53]str = ["verdict", "verdict", "ms", "count", "lines", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count", "verdict", "verdict", "ms", "count", "lines", "tokens", "count", "count"]; +pub const MEASUREMENT_EVIDENCE : [53]str = ["SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "SESSION-OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED", "OBSERVED"]; +pub const MEASUREMENT_SOURCES : [53]str = ["session-observed isolated worktree commands: current t27c v0.2.0 returned NOPARSE and test-report BLOCKED at unchanged line 37; no external run-log artifact was sealed", "session-observed Queen review of the control-arm diff and acceptance evidence on 2026-09-23; no external review artifact was sealed", "session-observed agent UTC timestamps 2026-09-23T04:10:25Z through 2026-09-23T04:32:27Z", "session-observed command ledger: unsupported tri flag and wrong bootstrap target; no external run-log artifact was sealed", "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 12 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile", "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree", "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt", "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration", "runtime-reported tool-use count of the arm's agent turn", "git diff --numstat in the arm's worktree: 18 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS", "runtime-reported token total of the arm's agent turn; input and output are not reported separately", "shell commands in the arm's transcript that invoked t27c", "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile"]; test protocol_has_one_variable { assert REAL_GITHUB_TASKS_ONLY == true; @@ -109,27 +113,46 @@ test protocol_has_one_variable { assert TRI_ROLE_EVIDENCE == "OBSERVED"; } -test arena_names_all_four_lanes { - assert CONFIG_COUNT == 4; +test arena_names_all_five_lanes { + assert CONFIG_COUNT == 5; assert CONFIG_IDS[0] == "bee-baseline"; - assert CONFIG_IDS[1] == "bee-jev"; - assert CONFIG_IDS[2] == "igla-coder"; - assert CONFIG_IDS[3] == "igla-race"; - assert CONFIG_EVIDENCE[1] == "SOURCE-CLAIM"; - assert CONFIG_STATE_EVIDENCE[1] == "OBSERVED"; + assert CONFIG_IDS[1] == "bee-tri"; + assert CONFIG_IDS[2] == "bee-jev"; + assert CONFIG_IDS[3] == "igla-coder"; + assert CONFIG_IDS[4] == "igla-race"; + assert CONFIG_EVIDENCE[1] == "OBSERVED"; + assert CONFIG_EVIDENCE[2] == "SOURCE-CLAIM"; + assert CONFIG_STATE_EVIDENCE[2] == "OBSERVED"; } test first_task_is_real_and_pinned { - assert EXPERIMENT_COUNT == 1; + assert EXPERIMENT_COUNT == 4; assert EXPERIMENT_ISSUE_NUMBERS[0] == "4328"; assert EXPERIMENT_REPOS[0] == "gHashTag/t27"; assert EXPERIMENT_STATES_BY_ID[0] == "credential-blocked"; assert EXPERIMENT_MODEL_EVIDENCE[0] == "UNKNOWN"; } +test campaign_20260926_is_paired_and_unranked { + assert EXPERIMENT_ISSUE_NUMBERS[1] == "4614"; + assert EXPERIMENT_ISSUE_NUMBERS[2] == "4695"; + assert EXPERIMENT_ISSUE_NUMBERS[3] == "4613"; + assert EXPERIMENT_STATES_BY_ID[3] == "judged"; + assert EXPERIMENT_MODEL_EVIDENCE[3] == "SESSION-OBSERVED"; + assert RUN_COUNT == 7; + assert RUN_CONFIG_IDS[1] == "bee-baseline"; + assert RUN_CONFIG_IDS[2] == "bee-tri"; + assert RUN_EXPERIMENT_IDS[1] == RUN_EXPERIMENT_IDS[2]; + assert RUN_PATCH_SHAS[1] == RUN_PATCH_SHAS[2]; + assert MEASUREMENT_COUNT == 53; +} + test metric_vocabulary_has_jev_and_cost_evidence { - assert METRIC_COUNT == 11; + assert METRIC_COUNT == 14; assert METRIC_KEYS[8] == "cost"; assert METRIC_KEYS[9] == "jev-latency"; assert METRIC_KEYS[10] == "jev-confidence"; + assert METRIC_KEYS[11] == "total-tokens"; + assert METRIC_KEYS[12] == "judge-calls"; + assert METRIC_KEYS[13] == "mutants-killed"; } diff --git a/apps/website/src/components/QueenRoadmap.tsx b/apps/website/src/components/QueenRoadmap.tsx index 3e6ce3fd36..b4470f014d 100644 --- a/apps/website/src/components/QueenRoadmap.tsx +++ b/apps/website/src/components/QueenRoadmap.tsx @@ -13,6 +13,7 @@ import { useEffect, useMemo, useState } from 'react' import './queenRoadmap.css' +import QueenRoadmapGame from './QueenRoadmapGame' type LangCount = { files: number; bytes: number } interface StackRepo { @@ -39,6 +40,11 @@ interface Goal { issue: number | null /** A GitHub issue search whose closed share is this stage's progress. */ progress?: string + /** Why nothing can be filed against this stage yet, when something stops it. */ + locked?: { en: string; ru: string } + /** Source the count above does not read (the endgame's upstream), measured by + the same rules; `source` names the commits and where they were read. */ + measured?: { bytes: number; at: string; source: string } } interface Goals { issueRepo: string @@ -81,7 +87,7 @@ const COPY = { title: 'ROADMAP', goal: 'The game: rewrite the whole stack in .t27', goalBody: - 'Everything below the interface is written once, in .t27, and generated to its target - Rust for servers, Zig, C and Verilog for the core and silicon. The one exception is the seed: t27c itself stays hand-written Rust. This tab counts how far the code that runs app.t27.ai is from that, and the plan to close it.', + 'Everything below the interface is written once, in .t27, and generated to its target - Rust for servers, Zig, C and Verilog for the core and silicon. The one exception is the seed: t27c itself stays hand-written Rust. In the end the rewrite reaches past our own code: the whole BrowserOS browser and every third-party dependency we run in another language. This tab counts how far the code that runs app.t27.ai is from that, and the plan to close it.', share: 'of the stack is .t27 today', toPort: 'still to rewrite', inT27: 'already in .t27', @@ -97,6 +103,7 @@ const COPY = { stateNone: 'no issue yet', stateUnknown: 'state unknown', target: 'target', + notMeasured: 'not measured yet', loading: 'Reading the count…', failed: 'The count could not be read.', joinTitle: 'The rewrite is one file at a time. Take one.', @@ -115,7 +122,7 @@ const COPY = { title: 'ДОРОЖНАЯ КАРТА', goal: 'Игра: переписать весь стек на .t27', goalBody: - 'Всё ниже интерфейса пишется один раз, на .t27, и генерируется в свою цель — Rust для серверов, Zig, C и Verilog для ядра и кремния. Единственное исключение — зерно: сам t27c остаётся рукописным Rust. Эта вкладка считает, насколько код, на котором работает app.t27.ai, далёк от этого, и показывает план, как дойти.', + 'Всё ниже интерфейса пишется один раз, на .t27, и генерируется в свою цель — Rust для серверов, Zig, C и Verilog для ядра и кремния. Единственное исключение — зерно: сам t27c остаётся рукописным Rust. В итоге переписывание выходит за пределы нашего кода: весь браузер BrowserOS и каждая сторонняя зависимость на другом языке. Эта вкладка считает, насколько код, на котором работает app.t27.ai, далёк от этого, и показывает план, как дойти.', share: 'стека уже на .t27', toPort: 'ещё переписать', inT27: 'уже на .t27', @@ -131,6 +138,7 @@ const COPY = { stateNone: 'задачи ещё нет', stateUnknown: 'состояние неизвестно', target: 'цель', + notMeasured: 'ещё не измерено', loading: 'Читаю подсчёт…', failed: 'Подсчёт прочитать не удалось.', joinTitle: 'Переписывание идёт по одному файлу. Возьмите один.', @@ -268,6 +276,9 @@ export default function QueenRoadmap({ lang }: { lang: 'en' | 'ru' }) { } return n } + // A stage measured outside the count uses that measurement alone: adding + // goalLines too would count stage 2's trios/agent-server a second time. + const stageBytes = (goal: Goal): number => goal.measured?.bytes ?? goalLines(goal) if (stack === null) return

{c.loading}

if (stack === 'failed' || !summary) return

{c.failed}

@@ -276,6 +287,18 @@ export default function QueenRoadmap({ lang }: { lang: 'en' | 'ru' }) { return (
+ {/* The game first, the measurement after: the comb is what the swarm is + doing to the numbers below. */} + {goals && ( + [g.stage, stageBytes(g)]))} + /> + )} +

{c.goal}

@@ -451,7 +474,11 @@ export default function QueenRoadmap({ lang }: { lang: 'en' | 'ru' }) { {l} ))} → {c.target}: {g.target} - {size(goalLines(g))} + {/* 0 bytes is "not in the count", not a size. A stage + measured outside the count names its source. */} + + {stageBytes(g) > 0 ? size(stageBytes(g)) : c.notMeasured} +
diff --git a/apps/website/src/components/QueenRoadmapGame.tsx b/apps/website/src/components/QueenRoadmapGame.tsx new file mode 100644 index 0000000000..578f9955a6 --- /dev/null +++ b/apps/website/src/components/QueenRoadmapGame.tsx @@ -0,0 +1,933 @@ +// LEVEL II: the rewrite, as the game it is. +// +// The ROADMAP tab measured the goal - bytes per language, a dial, eight stage +// cards - and never showed anyone PLAYING it. This is the board of that game: +// +// * THE COMB is an inverted pyramid, apex down, the shape of the t27 mark. +// The apex is the seed (t27c, the one hand-written thing everything else is +// generated from). Every other cell is one port task - one file of the +// stack becoming .t27 - and the cells are laid from the apex upward in the +// order the swarm should take them: the quickest first (fewest functions), +// so the comb is built from its point outward, the way the feeder files +// work. The widest row is the summit: the whole stack, and in the end the +// browser and every dependency we run, in .t27. +// * A BEE AT WORK is a ship over its cell with a beam into it. Only what the +// Queen's own board says is running is drawn as a ship; the page invents no +// activity, and when the board does not answer it draws no ships and says so. +// * THE QUEEN'S ROUND is a band of light that climbs the comb from the apex, +// timed by the board's own pulse (lastRoundAt, roundSeconds). +// * A CRACKED CELL is a built file with an open defect filed against it: the +// cell stays built and shows the crack until the defect closes. +// * THE RAID moves to another sector every UTC midnight; the roadmap feeder +// files that sector's tasks first the same day (gHashTag/t27 +// tools/queen/feed_roadmap.py, RAID_STAGES), and its cells built today +// count double HONEY - the board's own score, the functions ported. +// * BOSSES are stage 8 (the interface decision) and the endgame. Their HP is +// the measured source still to rewrite; a fight opens when half the comb +// below is built, and each says what really stands in the way. +// * BUILDERS are the lenders whose lanes built .t27 cells, read from the +// Queen's public leaderboard; XP is hers, never recomputed here. +// * THE COMB IS DRAWN IN 3D by default, in Babylon.js like the Queen's field +// (queenRoadmapScene.ts): the same cells as hex prisms on a wall, ships +// and the round included. Without WebGL, or by choice (Flat, or the +// field's own ?engine=canvas), it is the flat SVG below. +// +// Sources, all live and all public: GitHub issue search (port tasks open and +// closed as completed, and open defects naming a port file), the Queen's +// /queen/public-board and /queen/public-leaderboard. A source that does not +// answer is named on the page, and what it would have fed reads as unknown. + +import { useEffect, useMemo, useRef, useState } from 'react' +import { QUEEN_API } from '../lib/queenApi' +import { + BOSS_OPENS_AT, + builtShareBelow, + crackedTargets, + honeyOf, + msToNextDay, + parsePortTitle, + pulsePhase, + raidOf, + rowsFor, + targetOf, + utcDay, +} from '../lib/roadmapGame' +import type { RoadmapSceneCell, RoadmapSceneHandle, RoadmapSceneInput } from './queenRoadmapScene' +import './queenRoadmapGame.css' + +export interface GameGoal { + id: string + stage: number + title: { en: string; ru: string } + repos: string[] + languages: string[] + issue: number | null + locked?: { en: string; ru: string } +} + +type CellState = 'seed' | 'built' | 'cracked' | 'review' | 'building' | 'held' | 'target' | 'future' + +interface PortTask { + number: number + title: string + repo: string + path: string + lang: string + units: number + stage: number | null + state: CellState + closedAt: string | null +} + +interface Builder { + name: string + claimed: boolean + specs: number + xp: number +} + +const MARK = '#08fab5' + +const LANG_COLOR: Record = { + Python: '#4b8bbe', + Shell: '#89e051', + TypeScript: '#3178c6', + JavaScript: '#f1e05a', + Go: '#00add8', + Rust: '#dea584', + Zig: '#ec915c', + C: '#a8b9cc', + Verilog: '#b2b7f8', + Gleam: '#ffaff3', +} + +const COPY = { + en: { + level: 'LEVEL II · THE REWRITE', + title: 'The comb is built from its point up', + lead: + 'Every cell is one file of the stack becoming .t27. The point at the bottom is the seed, t27c; the bees build upward from the quickest tasks, row by row, to the widest row: the whole stack in .t27, and in the end the browser and every dependency we run.', + summit: 'Top row: the whole stack in .t27, the BrowserOS browser and every dependency included', + seed: 'The point: the seed, t27c, the one hand-written thing', + quick: 'quickest tasks first', + built: 'built', + cracked: 'cracked: a defect is open against it', + review: 'waiting for the Queen', + building: 'a bee is building', + held: 'files held', + target: 'target, free', + future: 'not filed yet', + raidCell: 'raid sector', + hudBuilt: 'cells built', + hudHoney: 'honey', + hudHoneyHint: + 'Honey is this board’s own score: the functions ported in built cells, with raid-sector cells built today counted twice. The Queen’s XP is on the leaderboard.', + hudBuilding: 'bees building now', + hudTargets: 'free targets', + hudCracked: 'cracked cells', + hudRow: 'highest row reached', + raid: 'Raid of the day', + raidBody: (title: string) => + `Sector ${title}. The roadmap feeder files its tasks first today, and its cells built today give double honey.`, + raidEnds: (h: number, m: number) => `The raid moves at UTC midnight, in ${h} h ${m} min.`, + raidNone: 'No sector can be raided: every stage below the bosses is locked.', + round: (sec: number, left: number) => `The Queen’s round: every ${sec} s, the next in ${left} s. The band of light is her round climbing the comb.`, + roundNone: 'The Queen’s round is not known: the board gave no pulse.', + targets: 'Priority targets: the quickest free cells', + targetsNone: 'No free port task right now. The roadmap feeder files the next ones every hour.', + attack: 'take it', + fn: 'fn', + sectors: 'Sectors', + captured: 'captured', + underAttack: 'under attack', + openFront: 'open front', + noTasks: 'no tasks filed yet', + locked: 'locked', + inFlight: 'in flight', + free: 'free', + unknown: 'unknown: GitHub did not answer', + raidTag: 'raid', + bosses: 'Bosses', + bossHp: 'HP: source still to rewrite', + bossHpUnknown: 'HP: not measured yet', + bossOpens: (pct: number) => + `The fight opens when ${Math.round(BOSS_OPENS_AT * 100)}% of the comb below is built (this board’s rule). Built now: ${pct}%.`, + bossOpen: 'The fight is open: half the comb below is built.', + bossWhy: 'What really stands in the way:', + builders: 'Builders: whose lanes built .t27 cells', + buildersBody: 'From the Queen’s public leaderboard: accepted issues whose boundary named a .t27 file, and her XP.', + buildersNone: 'No lane has built a .t27 cell yet.', + buildersDown: 'The Queen’s leaderboard did not answer.', + cells: 'cells', + unclaimed: 'unclaimed lane', + boardDown: "The Queen's board did not answer, so no ship is drawn: a ship is only what the board reports running.", + githubDown: 'GitHub did not answer the task search, so the comb shows only the seed and the frame.', + loading: 'Reading the comb…', + ships: 'Ships over the comb right now', + shipsNone: 'No bee is building a port cell right now.', + legend: 'Legend', + viewLabel: 'How the comb is drawn', + view3d: '3D', + view2d: 'Flat', + sceneHint: 'Drag to tilt the wall · point at a cell to lift it · click to open its issue', + sceneHintTouch: 'Tap a cell to see it, then open its issue', + sceneLost: (why: string) => `The 3D wall stopped (${why}), so the comb is drawn flat.`, + openIssue: 'open the issue', + }, + ru: { + level: 'УРОВЕНЬ II · ПЕРЕПИСЫВАНИЕ', + title: 'Соты строятся от вершины вверх', + lead: + 'Каждая сота — один файл стека, который становится .t27. Вершина внизу — зерно, t27c; пчёлы строят вверх от самых быстрых задач, ряд за рядом, до самого широкого ряда: весь стек на .t27, а в итоге и браузер, и каждая сторонняя зависимость, которой мы пользуемся.', + summit: 'Верхний ряд: весь стек на .t27, включая браузер BrowserOS и каждую зависимость', + seed: 'Вершина: зерно t27c, единственное рукописное', + quick: 'сначала самые быстрые задачи', + built: 'построено', + cracked: 'трещина: против неё открыт дефект', + review: 'ждёт Королеву', + building: 'пчела строит', + held: 'файлы заняты', + target: 'цель, свободна', + future: 'ещё не заведено', + raidCell: 'сектор рейда', + hudBuilt: 'сот построено', + hudHoney: 'мёд', + hudHoneyHint: + 'Мёд — собственный счёт этой доски: функции, перенесённые в построенных сотах; соты сектора рейда, построенные сегодня, считаются дважды. XP Королевы — в лидерборде.', + hudBuilding: 'пчёл строят сейчас', + hudTargets: 'свободных целей', + hudCracked: 'сот с трещиной', + hudRow: 'самый высокий ряд', + raid: 'Рейд дня', + raidBody: (title: string) => + `Сектор ${title}. Фидер роадмапа сегодня заводит его задачи первыми, а его соты, построенные сегодня, дают двойной мёд.`, + raidEnds: (h: number, m: number) => `Рейд сменится в полночь UTC, через ${h} ч ${m} мин.`, + raidNone: 'Рейдить некого: все этапы ниже боссов закрыты.', + round: (sec: number, left: number) => `Раунд Королевы: каждые ${sec} с, следующий через ${left} с. Полоса света — её раунд, поднимающийся по сотам.`, + roundNone: 'Раунд Королевы неизвестен: доска не дала пульса.', + targets: 'Приоритетные цели: самые быстрые свободные соты', + targetsNone: 'Свободной задачи переноса сейчас нет. Фидер роадмапа заводит следующие каждый час.', + attack: 'взять', + fn: 'фн', + sectors: 'Секторы', + captured: 'захвачен', + underAttack: 'под атакой', + openFront: 'открытый фронт', + noTasks: 'задач ещё нет', + locked: 'закрыт', + inFlight: 'в полёте', + free: 'свободно', + unknown: 'неизвестно: GitHub не ответил', + raidTag: 'рейд', + bosses: 'Боссы', + bossHp: 'HP: код, который ещё переписать', + bossHpUnknown: 'HP: ещё не измерено', + bossOpens: (pct: number) => + `Бой открывается, когда построено ${Math.round(BOSS_OPENS_AT * 100)}% сот ниже (правило этой доски). Сейчас построено: ${pct}%.`, + bossOpen: 'Бой открыт: половина сот ниже построена.', + bossWhy: 'Что на самом деле стоит на пути:', + builders: 'Строители: чьи полосы построили соты .t27', + buildersBody: 'Из публичного лидерборда Королевы: принятые задачи, чья граница называла файл .t27, и её XP.', + buildersNone: 'Ни одна полоса ещё не построила соту .t27.', + buildersDown: 'Лидерборд Королевы не ответил.', + cells: 'сот', + unclaimed: 'полосу никто не назвал своей', + boardDown: 'Доска Королевы не ответила, поэтому кораблей нет: корабль рисуется, только если доска говорит, что пчела работает.', + githubDown: 'GitHub не ответил на поиск задач, поэтому видны только зерно и контур.', + loading: 'Читаю соты…', + ships: 'Корабли над сотами прямо сейчас', + shipsNone: 'Сейчас ни одна пчела не строит соту переноса.', + legend: 'Обозначения', + viewLabel: 'Как нарисованы соты', + view3d: '3D', + view2d: 'Плоско', + sceneHint: 'Потяните, чтобы наклонить стену · наведите на соту, чтобы поднять её · нажмите, чтобы открыть задачу', + sceneHintTouch: 'Коснитесь соты, чтобы увидеть её, а затем откройте задачу', + sceneLost: (why: string) => `3D-стена остановилась (${why}), поэтому соты нарисованы плоско.`, + openIssue: 'открыть задачу', + }, +} as const + +function stageOf(goals: GameGoal[], repo: string, lang: string): number | null { + const both = goals.find((g) => g.repos.includes(repo) && g.languages.includes(lang)) + if (both) return both.stage + const byRepo = goals.find((g) => g.repos.includes(repo)) + return byRepo ? byRepo.stage : null +} + +const STATE_ORDER: Record = { + seed: 0, built: 1, cracked: 2, review: 3, building: 4, held: 5, target: 6, future: 7, +} + +type SearchRow = { number: number; title: string; closed_at?: string | null; pull_request?: unknown } + +async function search(q: string): Promise { + const r = await fetch( + `https://api.github.com/search/issues?per_page=100&sort=created&order=asc&q=${encodeURIComponent(q)}`, + { credentials: 'omit' }, + ) + if (!r.ok) throw new Error(String(r.status)) + const j = (await r.json()) as { items?: SearchRow[] } + return (j.items ?? []).filter((row) => !row.pull_request) +} + +const size = (bytes: number) => + bytes >= 1e6 ? `${(bytes / 1e6).toFixed(bytes >= 1e7 ? 0 : 1)} MB` : `${Math.max(1, Math.round(bytes / 1e3))} KB` + +export default function QueenRoadmapGame({ + lang, + goals, + issueRepo, + goalStates, + measuredBytes, +}: { + lang: 'en' | 'ru' + goals: GameGoal[] + issueRepo: string + goalStates: Record | null + /** Bytes of source still to rewrite per stage, from the count or the stage's own + measurement (goals.json `measured`); absent or 0 means not measured. */ + measuredBytes: Record +}) { + const c = lang === 'ru' ? COPY.ru : COPY.en + const [open, setOpen] = useState(null) + const [done, setDone] = useState(null) + const [defects, setDefects] = useState([]) + const [githubDown, setGithubDown] = useState(false) + const [board, setBoard] = useState | null>(null) + const [pulse, setPulse] = useState<{ lastRoundAt: string | null; roundSeconds: number } | null>(null) + const [boardDown, setBoardDown] = useState(false) + const [builders, setBuilders] = useState(null) + const [now, setNow] = useState(() => Date.now()) + // 3D is the default, as on the field; `?engine=canvas` (the field's own + // switch) or a stored choice draws the comb flat. + const [view, setView] = useState<'3d' | '2d'>(() => { + try { + if (new URLSearchParams(window.location.search).get('engine') === 'canvas') return '2d' + return window.localStorage.getItem('rg-view') === '2d' ? '2d' : '3d' + } catch { + return '3d' + } + }) + // Why the 3D wall could not be drawn, in the browser's words; the flat comb + // is shown instead. + const [sceneLost, setSceneLost] = useState(null) + const [hover, setHover] = useState<{ index: number; x: number; y: number; pinned: boolean } | null>(null) + const [reduced, setReduced] = useState(() => window.matchMedia('(prefers-reduced-motion: reduce)').matches) + const [finePointer] = useState(() => window.matchMedia('(pointer: fine)').matches) + useEffect(() => { + const query = window.matchMedia('(prefers-reduced-motion: reduce)') + const onChange = () => setReduced(query.matches) + query.addEventListener('change', onChange) + return () => query.removeEventListener('change', onChange) + }, []) + + useEffect(() => { + let live = true + const base = `repo:${issueRepo} is:issue in:title Port` + Promise.all([search(`${base} is:open`), search(`${base} is:closed reason:completed`)]) + .then(([o, d]) => { + if (!live) return + setOpen(o) + setDone(d) + }) + .catch(() => live && setGithubDown(true)) + // Defects against built port files. A failure here costs only the cracks, + // so it is not reported as GitHub being down. + search(`repo:${issueRepo} is:issue is:open in:title "specs/port"`) + .then((rows) => live && setDefects(rows)) + .catch(() => {}) + const readBoard = () => + fetch(`${QUEEN_API}/queen/public-board`, { credentials: 'omit' }) + .then((r) => (r.ok ? r.json() : Promise.reject(r.status))) + .then( + (raw: { + cards?: Array<{ number?: unknown; column?: unknown }> + pulse?: { lastRoundAt?: unknown; roundSeconds?: unknown } + }) => { + if (!live) return + const map: Record = {} + for (const card of raw.cards ?? []) { + if (typeof card.number === 'number' && typeof card.column === 'string') map[card.number] = card.column + } + setBoard(map) + const p = raw.pulse + setPulse( + p && typeof p.roundSeconds === 'number' + ? { lastRoundAt: typeof p.lastRoundAt === 'string' ? p.lastRoundAt : null, roundSeconds: p.roundSeconds } + : null, + ) + setBoardDown(false) + }, + ) + .catch(() => live && setBoardDown(true)) + readBoard() + fetch(`${QUEEN_API}/queen/public-leaderboard`, { credentials: 'omit' }) + .then((r) => (r.ok ? r.json() : Promise.reject(r.status))) + .then((raw: { contributors?: Array> }) => { + if (!live) return + const rows = (raw.contributors ?? []) + .map((row) => ({ + name: typeof row.name === 'string' ? row.name : '', + claimed: row.claimed === true, + specs: typeof row.specs === 'number' ? row.specs : 0, + xp: typeof row.xp === 'number' ? row.xp : 0, + })) + .filter((b) => b.specs > 0) + .sort((a, b) => b.specs - a.specs || b.xp - a.xp) + .slice(0, 5) + setBuilders(rows) + }) + .catch(() => live && setBuilders('down')) + const boardTimer = window.setInterval(readBoard, 30_000) + // One tick a second drives the round countdown and the raid clock. + const clock = window.setInterval(() => setNow(Date.now()), 1_000) + return () => { + live = false + window.clearInterval(boardTimer) + window.clearInterval(clock) + } + }, [issueRepo]) + + const cracked = useMemo(() => crackedTargets(defects.map((d) => d.title)), [defects]) + + const tasks = useMemo(() => { + const out: PortTask[] = [] + const add = (row: SearchRow, closed: boolean) => { + const parsed = parsePortTitle(row.title) + if (!parsed) return + const column = board?.[row.number] + const target = targetOf(row.title) + const state: CellState = closed + ? target && cracked.has(target) + ? 'cracked' + : 'built' + : column === 'running' + ? 'building' + : column === 'review' || column === 'done' + ? 'review' + : column === 'blocked' + ? 'held' + : 'target' + out.push({ + number: row.number, + title: row.title, + ...parsed, + stage: stageOf(goals, parsed.repo, parsed.lang), + state, + closedAt: row.closed_at ?? null, + }) + } + for (const row of done ?? []) add(row, true) + for (const row of open ?? []) add(row, false) + // The order the comb is built in: quickest first, then how far along. + return out.sort((a, b) => a.units - b.units || STATE_ORDER[a.state] - STATE_ORDER[b.state] || a.number - b.number) + }, [open, done, board, goals, cracked]) + + const day = utcDay(now) + const raid = raidOf(goals, day) + const raidGoal = goals.find((g) => g.stage === raid) ?? null + const toMidnight = msToNextDay(now) + const phase = pulse ? pulsePhase(pulse.lastRoundAt, pulse.roundSeconds, now) : null + const honey = honeyOf(tasks, raid, day) + const share = builtShareBelow(tasks) + + const rows = rowsFor(tasks.length + 1) + const capacity = (rows * (rows + 1)) / 2 + // Memoised, so the 3D wall is rebuilt only when the cells change, not on the + // clock's tick. + const shown = useMemo(() => tasks.slice(0, capacity - 1), [tasks, capacity]) + + // Geometry: pointy-top hexes, apex row at the bottom. + const s = 15 + const w = Math.sqrt(3) * s + // The frame's sides run at the comb's own slope, 1/sqrt(3) across per unit + // down, from an apex 2s below the seed: that clears every edge cell by 0.29 s. + const topY = s * 2.6 + const baseY = topY + (rows - 1) * 1.5 * s + s * 1.3 + const apexY = baseY + 2 * s + const frameTop = topY - s * 1.4 + const halfTop = (apexY - frameTop) / Math.sqrt(3) + const width = 2 * halfTop + 8 + const cx = width / 2 + const height = apexY + 6 + const hex = (x: number, y: number, r: number) => + Array.from({ length: 6 }, (_, k) => { + const a = (Math.PI / 180) * (60 * k - 90) + return `${(x + r * Math.cos(a)).toFixed(1)},${(y + r * Math.sin(a)).toFixed(1)}` + }).join(' ') + const framePoints = `${(cx - halfTop).toFixed(1)},${frameTop.toFixed(1)} ${(cx + halfTop).toFixed(1)},${frameTop.toFixed(1)} ${cx.toFixed(1)},${apexY.toFixed(1)}` + + type Placed = { x: number; y: number; row: number; task: PortTask | null; seed: boolean } + const placed = useMemo(() => { + const out: Placed[] = [] + let k = 0 + for (let r = 0; r < rows; r += 1) { + for (let i = 0; i <= r; i += 1) { + const x = cx + (i - r / 2) * w + const y = baseY - r * 1.5 * s + if (k === 0) out.push({ x, y, row: r, task: null, seed: true }) + else out.push({ x, y, row: r, task: shown[k - 1] ?? null, seed: false }) + k += 1 + } + } + return out + }, [shown, rows, cx, baseY, w]) + + // The same cells for the 3D wall (queenRoadmapScene.ts), in the same order. + const sceneCells = useMemo( + () => + placed.map((p) => ({ + x: p.x, + y: p.y, + kind: p.seed ? 'seed' : p.task ? p.task.state : 'future', + number: p.task?.number ?? null, + langHex: p.task ? LANG_COLOR[p.task.lang] ?? '#7a7f86' : MARK, + raid: raid !== null && p.task !== null && p.task.stage === raid, + })), + [placed, raid], + ) + const sceneInput = useMemo( + () => ({ + cells: sceneCells, + s, + frame: { cx, top: frameTop, apex: apexY, halfTop }, + pulse, + motion: reduced ? 'static' : 'interactive', + }), + [sceneCells, cx, frameTop, apexY, halfTop, pulse, reduced], + ) + const want3d = view === '3d' && sceneLost === null + const canvasRef = useRef(null) + const sceneRef = useRef(null) + const inputRef = useRef(sceneInput) + const cellsRef = useRef(sceneCells) + useEffect(() => { + inputRef.current = sceneInput + cellsRef.current = sceneCells + sceneRef.current?.update(sceneInput) + }, [sceneInput, sceneCells]) + // The wall is mounted once per switch to 3D and fed through update() after. + // Babylon is loaded only here, in its own chunk, the one the field uses. + useEffect(() => { + if (!want3d) return + const canvas = canvasRef.current + if (!canvas) return + let cancelled = false + import('./queenRoadmapScene') + .then(({ mountRoadmapScene }) => { + if (cancelled) return + try { + sceneRef.current = mountRoadmapScene(canvas, inputRef.current, { + onHover: (index, x, y) => + setHover((h) => (index === null ? (h?.pinned ? h : null) : { index, x, y, pinned: false })), + onPick: (index, pointerType, x, y) => { + const n = cellsRef.current[index]?.number + if (pointerType === 'mouse' && n) window.open(`https://github.com/${issueRepo}/issues/${n}`, '_blank', 'noopener,noreferrer') + else setHover({ index, x, y, pinned: true }) + }, + onLost: (why) => setSceneLost(why), + }) + } catch (e) { + setSceneLost(e instanceof Error ? e.message : String(e)) + } + }) + .catch((e) => { + if (!cancelled) setSceneLost(e instanceof Error ? e.message : String(e)) + }) + return () => { + cancelled = true + sceneRef.current?.dispose() + sceneRef.current = null + setHover(null) + } + }, [want3d, issueRepo]) + const chooseView = (v: '3d' | '2d') => { + setView(v) + if (v === '3d') setSceneLost(null) + try { + window.localStorage.setItem('rg-view', v) + } catch { + // Storage refused (a private window): the choice lasts this visit. + } + } + const hovered = hover ? placed[hover.index] ?? null : null + + const count = (state: CellState) => tasks.filter((t) => t.state === state).length + const builtCount = count('built') + count('cracked') + const highest = placed.reduce( + (m, p) => (p.task && (p.task.state === 'built' || p.task.state === 'cracked') ? Math.max(m, p.row + 1) : m), + 1, + ) + const building = placed.filter((p) => p.task?.state === 'building') + // The raid sector's free cells go first: that is what a raid is. + const targets = tasks + .filter((t) => t.state === 'target') + .sort((a, b) => Number(b.stage === raid) - Number(a.stage === raid) || a.units - b.units || a.number - b.number) + .slice(0, 8) + const stageTitle = (n: number | null) => goals.find((g) => g.stage === n)?.title[lang] ?? '' + const colorOfTask = (t: PortTask) => LANG_COLOR[t.lang] ?? '#7a7f86' + const loading = !githubDown && (open === null || done === null) + // Counts are shown only once the task search has answered. + const known = !githubDown && !loading + const pulseY = phase === null ? null : apexY - phase * (apexY - frameTop) + + const sector = (g: GameGoal) => { + const mine = tasks.filter((t) => t.stage === g.stage) + const b = mine.filter((t) => t.state === 'built' || t.state === 'cracked').length + const f = mine.filter((t) => t.state === 'building' || t.state === 'review').length + const o = mine.filter((t) => t.state === 'target' || t.state === 'held').length + const closed = g.issue !== null && goalStates?.[g.issue] === 'closed' + // Without the task search every count below is unknown, not zero. + const status = closed + ? c.captured + : githubDown && !g.locked + ? c.unknown + : g.locked && mine.length === 0 + ? `${c.locked}: ${g.locked[lang]}` + : f > 0 + ? c.underAttack + : o > 0 + ? c.openFront + : c.noTasks + const tone = closed ? 'cap' : g.locked && mine.length === 0 ? 'lock' : f > 0 ? 'hot' : o > 0 ? 'open' : 'none' + return { b, f, o, all: mine.length, status, tone } + } + + const bosses = goals.filter((g) => g.stage >= 8).sort((a, b) => a.stage - b.stage) + const maxBoss = Math.max(1, ...bosses.map((g) => measuredBytes[g.stage] ?? 0)) + const cellLabel = (t: PortTask) => + t.state === 'cracked' ? c.cracked : c[t.state as 'built' | 'review' | 'building' | 'held' | 'target'] + + return ( +
+
+ {c.level} +

{c.title}

+

{c.lead}

+
+ +
+
{known ? builtCount : '—'}{c.hudBuilt}
+
+ {known ? honey : '—'} + {c.hudHoney} ⓘ +
+
{known && board ? count('building') : '—'}{c.hudBuilding}
+
{known ? count('target') : '—'}{c.hudTargets}
+
{known ? count('cracked') : '—'}{c.hudCracked}
+
{known ? `${highest} / ${rows}` : '—'}{c.hudRow}
+
+ + + +
+
+
▽ {c.summit}
+
+ + +
+ {want3d ? ( +
+ + {hover && hovered && ( +
+ {hovered.seed || !hovered.task ? ( + <> + t27c + {c.seed} + + ) : ( + <> + + #{hovered.task.number} · {hovered.task.path} + + + {stageTitle(hovered.task.stage)} + {raid !== null && hovered.task.stage === raid ? ` · ${c.raidCell}` : ''} + + + {cellLabel(hovered.task)} · {hovered.task.units} {c.fn} + + {hover.pinned && ( + + {c.openIssue} → + + )} + + )} +
+ )} +
+ ) : ( + + + + + + + + + + + + + + + + + + + + + {/* The frame of the mark: the triangle every cell sits inside. */} + + + {/* The Queen's round, climbing from the apex; drawn under the cells. */} + {pulseY !== null && ( + + )} + + {placed.map((p, i) => { + if (p.seed) { + return ( + + + t27c + {c.seed} + + ) + } + if (!p.task) { + return + } + const t = p.task + const inRaid = raid !== null && t.stage === raid + return ( + + + {t.state === 'cracked' && ( + + )} + {t.state !== 'target' && t.state !== 'held' && t.state !== 'cracked' && ( + {t.units} + )} + {`#${t.number} · ${t.title}\n${stageTitle(t.stage)}${inRaid ? ` · ${c.raidCell}` : ''}\n${cellLabel(t)}`} + + ) + })} + + {/* Ships: one per cell the Queen's board says a bee is building. */} + {building.map((p, n) => { + const sx = p.x + (n % 2 === 0 ? -1 : 1) * w * 0.55 + const sy = p.y - s * 2.3 + return ( + + + + + + + + {`#${p.task!.number} · ${p.task!.title}`} + + ) + })} + + )} + {want3d &&

{finePointer ? c.sceneHint : c.sceneHintTouch}

} + {sceneLost && ( +

+ {c.sceneLost(sceneLost)} +

+ )} +
{c.seed} · {c.quick}
+

+ {pulse && phase !== null + ? c.round(pulse.roundSeconds, Math.max(0, Math.round((1 - phase) * pulse.roundSeconds))) + : boardDown + ? c.boardDown + : c.roundNone} +

+
+ + +
+ +

{c.targets}

+ {targets.length === 0 ? ( +

{githubDown ? c.githubDown : loading ? c.loading : c.targetsNone}

+ ) : ( +
    + {targets.map((t, n) => ( +
  1. + {n + 1} + + {t.stage === raid ? '⚔ ' : ''}{t.stage ? `S${t.stage}` : '·'} · {t.lang || '?'} + + + {t.repo === issueRepo ? '' : `${t.repo.replace('gHashTag/', '')}:`} + {t.path} + + {t.units} {c.fn} + + #{t.number} · {c.attack} → + +
  2. + ))} +
+ )} + +

{c.sectors}

+
    + {goals + .filter((g) => g.stage < 8) + .sort((a, b) => a.stage - b.stage) + .map((g) => { + const s2 = sector(g) + return ( +
  • + {g.stage} + + {g.title[lang]} + {g.stage === raid && ⚔ {c.raidTag}} + + {s2.status} + {s2.all > 0 && ( + + {s2.b} {c.built} · {s2.f} {c.inFlight} · {s2.o} {c.free} + + )} +
  • + ) + })} +
+ +

{c.bosses}

+
    + {bosses.map((g) => { + const hp = measuredBytes[g.stage] ?? 0 + const open2 = known && share >= BOSS_OPENS_AT + return ( +
  • +
    + {g.stage} + {g.title[lang]} +
    +
    0 ? `${c.bossHp}: ${size(hp)}` : c.bossHpUnknown}> + {hp > 0 ? `${c.bossHp}: ${size(hp)}` : c.bossHpUnknown} +
    +
    0 ? `${Math.max(4, (100 * hp) / maxBoss)}%` : '100%' }} className={hp > 0 ? '' : 'is-unknown'} /> +
    +
    +

    + {known ? (open2 ? c.bossOpen : c.bossOpens(Math.round(share * 100))) : c.unknown} +

    + {g.locked && ( +

    + {c.bossWhy} {g.locked[lang]} +

    + )} +
  • + ) + })} +
+
+ ) +} diff --git a/apps/website/src/components/QueenWars.css b/apps/website/src/components/QueenWars.css index 0fa6278815..ba3f128453 100644 --- a/apps/website/src/components/QueenWars.css +++ b/apps/website/src/components/QueenWars.css @@ -33,7 +33,9 @@ } .queen-wars-head, +.queen-wars-glance, .queen-wars-protocol, +.queen-wars-matrix, .queen-wars-mission, .queen-wars-arena, .queen-wars-score, @@ -110,6 +112,8 @@ .queen-wars-source code { margin-top: 5px; color: #fff5bd; } .queen-wars-protocol, +.queen-wars-glance, +.queen-wars-matrix, .queen-wars-mission, .queen-wars-arena, .queen-wars-score, @@ -277,7 +281,7 @@ .queen-wars-lanes { display: grid; - grid-template-columns: repeat(4, minmax(0, 1fr)); + grid-template-columns: repeat(5, minmax(0, 1fr)); } .queen-wars-lane { position: relative; @@ -437,6 +441,79 @@ white-space: normal; } +/* Campaign at a glance: four counts read before any detail. */ +.queen-wars-glance { padding: 14px clamp(16px, 2vw, 24px); } +.queen-wars-glance dl { + display: grid; + grid-template-columns: repeat(4, minmax(0, 1fr)); + gap: 1px; + margin: 0; + background: var(--wars-line); +} +.queen-wars-glance dl > div { + padding: 12px 14px; + background: rgba(1, 16, 13, 0.92); + -webkit-backdrop-filter: blur(6px); + backdrop-filter: blur(6px); +} +.queen-wars-glance dd { + margin: 0; + color: var(--wars-green); + /* 1.3, not 1.1: the self-hosted JetBrains Mono box is ~1.17em at weight 800, + and a shorter line counts as an undeclared scroller (queen-viewport-contract). */ + font: 800 clamp(24px, 3.2vw, 38px)/1.3 "JetBrains Mono", ui-monospace, monospace; + font-variant-numeric: tabular-nums; +} +.queen-wars-glance dt { margin-top: 6px; color: var(--wars-text-muted); font-size: 11px; } +.queen-wars-glance > p { margin: 10px 0 0; color: var(--wars-session); font-size: 12px; line-height: 1.5; } + +/* Head to head: every experiment in one grid; a row header opens it below. */ +.queen-wars-matrix { padding: 0; } +.queen-wars-matrix > .queen-wars-section-title { padding: 16px 20px 0; } +.queen-wars .queen-wars-matrix > p.queen-wars-matrix-hint { max-width: none; margin: 8px 20px 14px; color: var(--wars-text-muted); font-size: 12px; } +.queen-wars-matrix table { min-width: 640px; } +.queen-wars-matrix tbody th { padding: 0; } +.queen-wars-matrix tbody th button { + display: grid; + gap: 3px; + width: 100%; + min-height: 44px; + padding: 10px 16px; + border: 0; + background: transparent; + color: #e8fff7; + cursor: pointer; + text-align: left; +} +.queen-wars-matrix tbody th button:hover { background: rgba(0, 245, 160, 0.06); } +.queen-wars-matrix tbody th button:focus-visible { outline: 2px solid var(--wars-green); outline-offset: -2px; } +.queen-wars-matrix tbody th small { margin-top: 0; } +.queen-wars-matrix tr[aria-current="true"] th button { box-shadow: inset 3px 0 0 var(--wars-gold); background: rgba(255, 213, 74, 0.06); } +.queen-wars-matrix-empty { color: var(--wars-text-subtle); font-size: 11px; } + +/* A value's share of the largest value in its scoreboard row. */ +.queen-wars-bar { + display: block; + width: calc(var(--bar, 0) * 100%); + min-width: 2px; + height: 4px; + margin-top: 6px; + background: var(--wars-cyan); + opacity: 0.8; +} +.queen-wars-unmeasured { + display: flex; + flex-wrap: wrap; + align-items: center; + gap: 8px; + margin: 0; + padding: 12px 20px; + border-top: 1px solid var(--wars-line); + color: var(--wars-text-subtle); + font: 500 11px/1.5 "JetBrains Mono", ui-monospace, monospace; +} +.queen-wars-unmeasured .queen-wars-evidence { margin: 0; } + .queen-wars-empty { margin: 18px 0 0; color: var(--wars-text-muted); } .queen-wars-ledger ol { display: grid; gap: 8px; list-style: none; padding: 0; margin: 18px 0 0; overflow-x: auto; } .queen-wars-ledger li { display: grid; grid-template-columns: 160px 1fr auto auto; align-items: center; gap: 8px 16px; padding: 12px; border: 1px solid rgba(0, 245, 160, 0.15); } @@ -497,6 +574,7 @@ .queen-wars { padding: 10px; } .queen-wars-head { grid-template-columns: 1fr; align-items: stretch; } .queen-wars-source { width: 100%; } + .queen-wars-glance dl { grid-template-columns: repeat(2, minmax(0, 1fr)); } .queen-wars-protocol dl { grid-template-columns: 1fr; } .queen-wars-protocol dl > div { display: grid; gap: 4px; } .queen-wars-experiment-picker { grid-template-columns: 1fr; } diff --git a/apps/website/src/components/QueenWars.tsx b/apps/website/src/components/QueenWars.tsx index f783aa83a4..8b38f8cb03 100644 --- a/apps/website/src/components/QueenWars.tsx +++ b/apps/website/src/components/QueenWars.tsx @@ -1,4 +1,4 @@ -import { useState } from 'react' +import { useState, type CSSProperties } from 'react' import { QUEEN_WARS } from '../lib/queenWars.generated' import './QueenWars.css' @@ -58,12 +58,25 @@ const COPY = { ledger: 'RUN LEDGER', noLedger: 'No completed arm has been sealed into the .t27 ledger yet.', pipeline: 'TRAINING PIPELINE', - pipelineCopy: 'IGLA enters this arena only after an executable checkpoint exists. Until then its pipeline is visible, but it has no benchmark score.', + pipelineCopy: 'IGLA enters this arena when one of its checkpoints runs an arena issue under the same gates. The pilot reports trained checkpoints; none has run an arena issue yet, so IGLA has no arena score.', control: 'CONTROL', - decision: 'DECISION LAYER', + triLayer: 'TRI DECISION LAYER', + decision: 'COMPARISON ARM', ownModel: 'OWN MODEL TARGET', training: 'TRAINING RACE', specHash: 'spec sha256', + glance: 'CAMPAIGN AT A GLANCE', + statIssues: 'real issues', + statRuns: 'sealed runs', + statAccepted: 'accepted by the gates', + statMeasured: 'measurements', + statNoWinner: 'No winner is declared: the executor model id is not recorded in the ledger.', + matrix: 'HEAD TO HEAD', + matrixHint: 'One cell is one arm on one real issue. Choose a row to open it below.', + open: 'open', + noArmRun: 'no run', + notMeasured: 'not measured in this experiment', + mutantsKilled: 'mutants killed', }, ru: { kicker: 'АРЕНА АГЕНТОВ НА РЕАЛЬНЫХ ЗАДАЧАХ', @@ -96,12 +109,25 @@ const COPY = { ledger: 'ЖУРНАЛ ЗАПУСКОВ', noLedger: 'Ни одна завершённая рука ещё не запечатана в журнале .t27.', pipeline: 'КОНВЕЙЕР ОБУЧЕНИЯ', - pipelineCopy: 'IGLA входит на эту арену только после появления исполняемого checkpoint. До этого конвейер виден, но benchmark-оценки у него нет.', + pipelineCopy: 'IGLA входит на эту арену, когда один из её checkpoint решает issue арены под теми же воротами. Пилот сообщает об обученных checkpoint, но ни один ещё не решал issue арены, поэтому оценки на арене у IGLA нет.', control: 'КОНТРОЛЬ', - decision: 'СЛОЙ РЕШЕНИЙ', + triLayer: 'СЛОЙ РЕШЕНИЙ TRI', + decision: 'СРАВНИТЕЛЬНОЕ ПЛЕЧО', ownModel: 'ЦЕЛЬ: СВОЯ МОДЕЛЬ', training: 'ГОНКА ОБУЧЕНИЯ', specHash: 'sha256 спеки', + glance: 'КАМПАНИЯ ОДНИМ ВЗГЛЯДОМ', + statIssues: 'реальных issue', + statRuns: 'запечатанных запусков', + statAccepted: 'приняты воротами', + statMeasured: 'измерений', + statNoWinner: 'Победитель не объявлен: в журнале не записан id модели исполнителя.', + matrix: 'ЛИЦОМ К ЛИЦУ', + matrixHint: 'Одна клетка — одна рука на одной реальной issue. Выберите строку, чтобы открыть её ниже.', + open: 'открыть', + noArmRun: 'нет запуска', + notMeasured: 'не измерено в этом эксперименте', + mutantsKilled: 'поймано мутантов', }, } as const @@ -112,6 +138,7 @@ const STATE_RU: Record = { 'pipeline-only': 'только конвейер', planned: 'запланирован', running: 'выполняется', + judged: 'оценён, без победителя', complete: 'завершён', invalid: 'недействителен', passed: 'пройден', @@ -121,6 +148,7 @@ const STATE_RU: Record = { const kindLabel = (id: string, c: typeof COPY.en | typeof COPY.ru) => { if (id === 'bee-baseline') return c.control + if (id === 'bee-tri') return c.triLayer if (id === 'bee-jev') return c.decision if (id === 'igla-coder') return c.ownModel return c.training @@ -140,6 +168,12 @@ function Evidence({ value }: { value: string }) { return {value} } +// Units whose values compare on one axis within a row; a bar shows each arm's +// value relative to the largest in that row. Verdicts and probabilities do not. +const BAR_UNITS = new Set(['ms', 'tokens', 'count', 'lines', 'usd']) +// The arms a head-to-head row shows: the paired arms, then the comparison arm. +const MATRIX_ARMS = ['bee-baseline', 'bee-tri', 'bee-jev'] as const + export function QueenWars({ lang }: { lang: 'en' | 'ru' }) { const c = COPY[lang] const [selectedExperimentId, setSelectedExperimentId] = useState(QUEEN_WARS.experiments[0].id) @@ -153,6 +187,27 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) { const run = runFor(configId) return run ? measurements.find((item) => item.runId === run.id && item.key === key) : undefined } + const latestRun = (experimentId: string, configId: string) => + [...runs].reverse().find((run) => run.experimentId === experimentId && run.configId === configId) + + // Scoreboard: only the arms that ran this experiment, and only the metrics + // at least one of them measured. The rest are listed, never shown as zero. + const armsWithRuns = QUEEN_WARS.configurations.filter((config) => runFor(config.id)) + const scoreArms = armsWithRuns.length ? armsWithRuns : QUEEN_WARS.configurations + const measuredMetrics = QUEEN_WARS.metricCatalog.filter((metric) => scoreArms.some((config) => measureFor(config.id, metric.key))) + const unmeasuredMetrics = QUEEN_WARS.metricCatalog.filter((metric) => !measuredMetrics.includes(metric)) + const rowMax = (key: string) => Math.max(0, ...scoreArms.map((config) => Number(measureFor(config.id, key)?.value)).filter(Number.isFinite)) + + const sealedRuns = runs.filter((run) => run.evidence === 'OBSERVED' && run.artifactUrl && run.logSha256 && run.patchSha256) + const glance = [ + { value: QUEEN_WARS.experiments.length, label: c.statIssues }, + { value: sealedRuns.length, label: c.statRuns }, + { value: `${runs.filter((run) => run.verdict === 'accepted').length}/${runs.length}`, label: c.statAccepted }, + { value: measurements.length, label: c.statMeasured }, + ] + const unranked = QUEEN_WARS.experiments.some((item) => item.state === 'judged') + // A column no experiment has a run for would be a column of "no run". + const matrixArms = MATRIX_ARMS.filter((id) => runs.some((run) => run.configId === id)) return (
@@ -168,6 +223,18 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {
+
+
+ {glance.map((item) => ( +
+
{item.value}
+
{item.label}
+
+ ))} +
+ {unranked &&

{c.statNoWinner}

} +
+
01 @@ -198,9 +265,55 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {
+
+
+ 02 +

{c.matrix}

+
+

{c.matrixHint}

+
+ + + + + {matrixArms.map((id) => ( + + ))} + + + + + {QUEEN_WARS.experiments.map((item) => ( + + + {matrixArms.map((id) => { + const run = latestRun(item.id, id) + const killed = run ? measurements.find((m) => m.runId === run.id && m.key === 'mutants-killed') : undefined + return ( + + ) + })} + + + ))} + +
{c.task}{QUEEN_WARS.configurations.find((config) => config.id === id)?.name ?? id}{c.state}
+ + + {run ? <> + {run.verdict} + {state(run.state)}{killed ? ` · ${c.mutantsKilled}: ${killed.value}` : ''} + : {c.noArmRun}} + {state(item.state)}
+
+
+
- 02 + 03

{c.task}

@@ -240,7 +353,7 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {
- 03 + 04

{c.configurations}

@@ -286,7 +399,7 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {
- 04 + 05

{c.metrics}

@@ -294,15 +407,17 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) { {c.metric} - {QUEEN_WARS.configurations.map((config) => {config.name})} + {scoreArms.map((config) => {config.name})} - {QUEEN_WARS.metricCatalog.map((metric) => ( + {measuredMetrics.map((metric) => ( {metric.key}{metric.unit} - {QUEEN_WARS.configurations.map((config) => { + {scoreArms.map((config) => { const value = measureFor(config.id, metric.key) + const max = rowMax(metric.key) + const bar = value && BAR_UNITS.has(value.unit) && max > 0 ? Number(value.value) / max : null return ( {value.value} {value.unit} + {bar !== null &&
+ {unmeasuredMetrics.length > 0 && ( +

+ + {c.notMeasured}: {unmeasuredMetrics.map((metric) => metric.key).join(' · ')} +

+ )}
- 05 + 06

{c.ledger}

{experimentRuns.length === 0 ?

{c.noLedger}

: ( @@ -350,7 +472,7 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {
- 06 + 07

{c.pipeline}

{c.pipelineCopy}

diff --git a/apps/website/src/components/queenRoadmapGame.css b/apps/website/src/components/queenRoadmapGame.css new file mode 100644 index 0000000000..1d79bab1a2 --- /dev/null +++ b/apps/website/src/components/queenRoadmapGame.css @@ -0,0 +1,692 @@ +/* LEVEL II: the comb, apex down like the t27 mark, and the ships building it. */ +.rg { + --rg-mark: #08fab5; + --rg-gold: #ffd35a; + --rg-panel: rgba(2, 10, 8, 0.92); + --rg-line: rgba(8, 250, 181, 0.24); + --rg-text: #e8f5ee; + --rg-muted: rgba(232, 245, 238, 0.8); + margin: 0 0 28px; + padding: 18px; + border: 1px solid var(--rg-line); + border-radius: 14px; + background: + radial-gradient(ellipse at 50% 0%, rgba(8, 250, 181, 0.1), transparent 60%), + var(--rg-panel); + color: var(--rg-text); +} +.rg-head h2 { + margin: 4px 0 6px; + color: var(--rg-mark); +} +.rg-head p { + max-width: 820px; +} +.rg-level { + display: inline-block; + padding: 2px 8px; + border: 1px solid var(--rg-mark); + border-radius: 999px; + color: var(--rg-mark); + font-size: 11px; + letter-spacing: 0.16em; +} + +.rg-hud { + display: grid; + grid-template-columns: repeat(auto-fit, minmax(96px, 1fr)); + gap: 10px; + margin: 16px 0 12px; +} +.rg-hud > div { + padding: 10px 12px; + border: 1px solid var(--rg-line); + border-radius: 10px; + background: rgba(0, 0, 0, 0.35); +} +.rg-hud strong { + display: block; + font-size: 22px; + line-height: 1.15; + color: var(--rg-gold); + font-variant-numeric: tabular-nums; +} +.rg-hud span { + font-size: 12px; + color: var(--rg-muted); +} + +.rg-stage { + display: grid; + grid-template-columns: minmax(0, 1fr) 260px; + gap: 16px; + align-items: start; +} +/* Sized by the element, not by the svg's own attributes: the board styles every + svg inside its viewport. */ +.rg-comb { + display: block; + width: 100%; + height: auto; + max-height: 620px; +} +.rg-frame { + fill: rgba(8, 250, 181, 0.03); + stroke: rgba(8, 250, 181, 0.35); + stroke-width: 1.2; + stroke-linejoin: round; +} +.rg-fig { + margin: 0; + min-width: 0; +} +.rg-cap { + font-size: 12px; + letter-spacing: 0.04em; + text-align: center; + color: var(--rg-muted); +} +.rg-cap-top { + color: var(--rg-gold); + margin-bottom: 4px; +} +.rg-cap-bottom { + margin-top: 2px; +} + +.rg-cell polygon { + stroke-width: 1.3; + transition: filter 0.2s ease; +} +.rg-cell text { + font-size: 9px; + font-weight: 700; + pointer-events: none; +} +a.rg-cell:hover polygon, +a.rg-cell:focus-visible polygon { + filter: brightness(1.35) drop-shadow(0 0 4px var(--rg-mark)); +} +.rg-future { + fill: none; + stroke: rgba(8, 250, 181, 0.13); + stroke-width: 1; +} +.rg-seed polygon { + fill: var(--rg-mark); + stroke: #eafff7; + filter: drop-shadow(0 0 6px var(--rg-mark)); +} +.rg-seed text { + fill: #02140d; + font-size: 7.5px; +} +.rg-target polygon { + fill: color-mix(in srgb, var(--rg-lang, #7a7f86) 16%, transparent); + stroke: var(--rg-lang, #7a7f86); +} +.rg-held polygon { + fill: rgba(255, 107, 107, 0.1); + stroke: #ff8080; + stroke-dasharray: 3 2; +} +.rg-review polygon { + fill: rgba(255, 211, 90, 0.22); + stroke: var(--rg-gold); + stroke-dasharray: 4 2; +} +.rg-review text { + fill: var(--rg-gold); +} +.rg-building polygon { + fill: rgba(8, 250, 181, 0.22); + stroke: var(--rg-mark); + stroke-dasharray: 5 3; + animation: rg-scaffold 1.2s linear infinite; +} +.rg-building text { + fill: var(--rg-mark); +} +.rg-built polygon { + fill: url(#rg-honey); + stroke: var(--rg-mark); +} +.rg-built text { + fill: #2a1a00; +} + +.rg-ship-group { + animation: rg-bob 2.6s ease-in-out infinite; + transform-box: fill-box; +} +.rg-ship { + fill: #dffcf3; + stroke: var(--rg-mark); + stroke-width: 0.8; +} +.rg-cockpit { + fill: var(--rg-mark); +} +.rg-flame { + fill: #ffae3a; + transform-box: fill-box; + transform-origin: 50% 100%; + animation: rg-flame 0.25s ease-in-out infinite alternate; +} +.rg-beam { + animation: rg-beam 1.4s ease-in-out infinite; +} + +@keyframes rg-scaffold { + to { + stroke-dashoffset: -16; + } +} +@keyframes rg-bob { + 0%, + 100% { + transform: translateY(0); + } + 50% { + transform: translateY(-3px); + } +} +@keyframes rg-flame { + from { + transform: scaleY(0.7); + opacity: 0.75; + } + to { + transform: scaleY(1.15); + opacity: 1; + } +} +@keyframes rg-beam { + 0%, + 100% { + opacity: 0.45; + } + 50% { + opacity: 1; + } +} +@media (prefers-reduced-motion: reduce) { + .rg-building polygon, + .rg-ship-group, + .rg-flame, + .rg-beam { + animation: none; + } +} + +.rg-side h3, +.rg-h { + margin: 14px 0 8px; + font-size: 12px; + letter-spacing: 0.14em; + text-transform: uppercase; + color: var(--rg-gold); +} +.rg-side h3:first-child { + margin-top: 0; +} +.rg-legend, +.rg-ships { + list-style: none; + margin: 0; + padding: 0; + display: grid; + gap: 6px; + font-size: 13px; + color: var(--rg-muted); +} +.rg-legend li { + display: flex; + align-items: center; + gap: 8px; +} +.rg-ships li { + overflow-wrap: anywhere; +} +.rg-ships a, +.rg-targets a { + color: var(--rg-mark); +} +.rg-k { + width: 14px; + height: 14px; + flex: none; + clip-path: polygon(50% 0, 100% 25%, 100% 75%, 50% 100%, 0 75%, 0 25%); +} +.rg-k-built { + background: linear-gradient(#ffe28a, #e0a100); +} +.rg-k-review { + background: rgba(255, 211, 90, 0.55); +} +.rg-k-building { + background: var(--rg-mark); +} +.rg-k-target { + background: #3178c6; +} +.rg-k-held { + background: #ff8080; +} +.rg-k-future { + background: rgba(8, 250, 181, 0.2); +} +.rg-note { + font-size: 13px; + color: var(--rg-muted); +} + +.rg-targets { + list-style: none; + margin: 0; + padding: 0; + display: grid; + gap: 6px; +} +.rg-targets li { + display: grid; + grid-template-columns: 26px auto minmax(0, 1fr) auto auto; + gap: 10px; + align-items: center; + padding: 8px 10px; + border: 1px solid var(--rg-line); + border-radius: 10px; + background: rgba(0, 0, 0, 0.3); + font-size: 13px; +} +.rg-rank { + width: 22px; + height: 22px; + display: grid; + place-items: center; + border-radius: 50%; + background: var(--rg-mark); + color: #02140d; + font-weight: 700; + font-size: 12px; +} +.rg-chip { + padding: 2px 7px; + border-radius: 999px; + border: 1px solid var(--rg-lang, #7a7f86); + color: var(--rg-text); + font-size: 11px; + white-space: nowrap; +} +.rg-path { + min-width: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; + font-family: ui-monospace, SFMono-Regular, Menlo, monospace; + font-size: 12px; +} +.rg-units { + color: var(--rg-gold); + white-space: nowrap; +} +.rg-targets a { + white-space: nowrap; + font-weight: 600; +} + +.rg-sectors { + list-style: none; + margin: 0; + padding: 0; + display: grid; + grid-template-columns: repeat(auto-fill, minmax(230px, 1fr)); + gap: 8px; +} +.rg-sector { + display: grid; + grid-template-columns: 30px minmax(0, 1fr); + gap: 2px 10px; + padding: 10px; + border: 1px solid var(--rg-line); + border-radius: 10px; + background: rgba(0, 0, 0, 0.3); + font-size: 13px; +} +.rg-sector b { + color: var(--rg-text); +} +.rg-sector-num { + grid-row: span 3; + width: 28px; + height: 28px; + display: grid; + place-items: center; + clip-path: polygon(50% 0, 100% 25%, 100% 75%, 50% 100%, 0 75%, 0 25%); + background: rgba(8, 250, 181, 0.25); + color: var(--rg-text); + font-weight: 700; +} +.rg-sector-status { + font-size: 12px; + color: var(--rg-muted); +} +.rg-sector-count { + font-size: 12px; + color: var(--rg-muted); +} +.rg-tone-cap .rg-sector-num { + background: var(--rg-gold); + color: #2a1a00; +} +.rg-tone-hot { + border-color: var(--rg-mark); +} +.rg-tone-hot .rg-sector-num { + background: var(--rg-mark); + color: #02140d; +} +.rg-tone-hot .rg-sector-status { + color: var(--rg-mark); +} +.rg-tone-open .rg-sector-status { + color: #9fd0ff; +} +.rg-tone-lock { + opacity: 0.85; +} +.rg-tone-lock .rg-sector-num { + background: rgba(255, 128, 128, 0.3); +} + +/* Honey, the board's own score. */ +.rg-hud .rg-honey-num { + color: #ffc93c; +} +.rg-hud > div[title] { + cursor: help; +} + +/* The raid of the day. */ +.rg-raid { + display: flex; + flex-wrap: wrap; + gap: 4px 10px; + align-items: baseline; + margin: 0 0 12px; + padding: 10px 12px; + border: 1px solid rgba(255, 128, 64, 0.6); + border-radius: 10px; + background: rgba(40, 12, 2, 0.85); + color: var(--rg-text); + font-size: 13px; +} +.rg-raid b { + color: #ffae3a; + letter-spacing: 0.08em; + text-transform: uppercase; + font-size: 12px; +} +.rg-raid.is-none { + border-color: var(--rg-line); + background: rgba(0, 0, 0, 0.35); +} +.rg-in-raid.rg-target polygon { + stroke: #ffae3a; + stroke-width: 1.8; + animation: rg-raid-glow 1.6s ease-in-out infinite; +} +@keyframes rg-raid-glow { + 0%, + 100% { + filter: drop-shadow(0 0 0 transparent); + } + 50% { + filter: drop-shadow(0 0 4px #ffae3a); + } +} +.rg-targets li.is-raid { + border-color: rgba(255, 128, 64, 0.7); +} +.rg-sector.is-raid { + border-color: #ffae3a; +} +.rg-raid-tag { + margin-left: 6px; + padding: 0 6px; + border-radius: 999px; + background: #ffae3a; + color: #2a1200; + font-size: 11px; + font-weight: 700; + white-space: nowrap; +} + +/* The Queen's round, climbing the comb. */ +.rg-pulse { + pointer-events: none; +} +.rg-round { + margin-top: 4px; + font-size: 12px; + text-align: center; + color: var(--rg-muted); +} + +/* A cracked cell: built, with an open defect against it. */ +.rg-cracked polygon { + fill: url(#rg-honey); + stroke: #ff6b6b; + stroke-width: 1.8; + filter: saturate(0.55) brightness(0.8); +} +.rg-crack { + fill: none; + stroke: #2a0a0a; + stroke-width: 1.6; + stroke-linejoin: round; + pointer-events: none; +} +.rg-k-cracked { + background: linear-gradient(135deg, #c99a2e 45%, #2a0a0a 47%, #2a0a0a 53%, #c99a2e 55%); + outline: 1px solid #ff6b6b; +} +.rg-k-raid { + background: #ffae3a; +} + +/* Builders, from the Queen's leaderboard. */ +.rg-builders { + list-style: decimal inside; + margin: 0; + padding: 0; + display: grid; + gap: 4px; + font-size: 13px; + color: var(--rg-muted); +} +.rg-builders li { + display: grid; + grid-template-columns: minmax(0, 1fr) auto auto; + gap: 8px; +} +.rg-builder-name { + min-width: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; + color: var(--rg-text); +} +.rg-builder-cells { + color: var(--rg-gold); +} +.rg-builder-xp { + color: var(--rg-mark); +} + +/* Bosses. */ +.rg-bosses { + list-style: none; + margin: 0; + padding: 0; + display: grid; + grid-template-columns: repeat(auto-fill, minmax(260px, 1fr)); + gap: 10px; +} +.rg-boss { + display: grid; + gap: 8px; + padding: 12px; + border: 1px solid rgba(255, 107, 107, 0.45); + border-radius: 12px; + background: rgba(24, 4, 6, 0.85); + font-size: 13px; +} +.rg-boss.is-open { + border-color: #ff6b6b; + box-shadow: 0 0 12px rgba(255, 107, 107, 0.35); +} +.rg-boss-head { + display: flex; + gap: 10px; + align-items: center; +} +.rg-boss-head b { + color: var(--rg-text); +} +.rg-boss-num { + width: 30px; + height: 30px; + flex: none; + display: grid; + place-items: center; + clip-path: polygon(50% 0, 100% 25%, 100% 75%, 50% 100%, 0 75%, 0 25%); + background: #ff6b6b; + color: #2a0a0a; + font-weight: 800; +} +.rg-boss-hp span { + display: block; + font-size: 12px; + color: var(--rg-muted); + margin-bottom: 4px; +} +.rg-boss-bar { + height: 10px; + border-radius: 999px; + background: rgba(255, 255, 255, 0.08); + overflow: hidden; +} +.rg-boss-bar div { + height: 100%; + background: linear-gradient(90deg, #ff6b6b, #ffae3a); +} +.rg-boss-bar div.is-unknown { + background: repeating-linear-gradient(45deg, rgba(255, 107, 107, 0.35) 0 6px, transparent 6px 12px); +} + +@media (prefers-reduced-motion: reduce) { + .rg-in-raid.rg-target polygon { + animation: none; + } +} + +/* The comb as a 3D wall (queenRoadmapScene.ts), drawn like the field: a hive + hanging in the dark with a few stars behind it. The canvas is transparent; + this box is the sky. */ +.rg-scene { + position: relative; + width: 100%; + max-height: 620px; + border-radius: 10px; + overflow: hidden; + background: + radial-gradient(1.2px 1.2px at 12% 22%, rgba(234, 255, 247, 0.7) 50%, transparent 55%), + radial-gradient(1px 1px at 27% 64%, rgba(234, 255, 247, 0.5) 50%, transparent 55%), + radial-gradient(1.4px 1.4px at 71% 14%, rgba(255, 211, 90, 0.6) 50%, transparent 55%), + radial-gradient(1px 1px at 86% 48%, rgba(234, 255, 247, 0.55) 50%, transparent 55%), + radial-gradient(1px 1px at 58% 83%, rgba(234, 255, 247, 0.45) 50%, transparent 55%), + radial-gradient(1.2px 1.2px at 40% 9%, rgba(100, 220, 255, 0.55) 50%, transparent 55%), + radial-gradient(120% 80% at 50% 105%, rgba(8, 250, 181, 0.12), transparent 60%), + #010706; +} +.rg-scene-canvas { + display: block; + width: 100%; + height: 100%; + outline: none; + touch-action: pan-y; +} +.rg-hover { + position: absolute; + z-index: 2; + display: grid; + gap: 2px; + max-width: 250px; + padding: 8px 10px; + border: 1px solid var(--rg-line); + border-radius: 8px; + background: var(--rg-panel); + /* Read while the wall moves under it: the comb's edges are blurred out. */ + backdrop-filter: blur(8px); + color: var(--rg-text); + font-size: 12px; + line-height: 1.35; + transform: translate(12px, 12px); + pointer-events: none; + overflow-wrap: anywhere; +} +.rg-hover b { + color: var(--rg-mark); +} +.rg-hover.is-pinned { + pointer-events: auto; +} +.rg-hover a { + color: var(--rg-mark); +} +.rg-view { + display: flex; + justify-content: center; + gap: 4px; + margin: 0 0 6px; +} +.rg-view button { + font: inherit; + font-size: 11px; + padding: 2px 12px; + border: 1px solid var(--rg-line); + border-radius: 999px; + background: var(--rg-panel); + backdrop-filter: blur(6px); + color: var(--rg-text); + cursor: pointer; +} +.rg-view button[aria-pressed='true'] { + background: var(--rg-mark); + border-color: var(--rg-mark); + color: #02140d; +} +.rg-scene-hint, +.rg-scene-note { + margin: 4px 0 0; + font-size: 12px; + text-align: center; + color: var(--rg-muted); +} + +@media (max-width: 760px) { + .rg { + padding: 12px; + } + .rg-stage { + grid-template-columns: minmax(0, 1fr); + } + .rg-targets li { + grid-template-columns: 22px minmax(0, 1fr) auto; + } + .rg-targets li .rg-chip, + .rg-targets li .rg-units { + display: none; + } +} diff --git a/apps/website/src/components/queenRoadmapScene.ts b/apps/website/src/components/queenRoadmapScene.ts new file mode 100644 index 0000000000..9dfedc56ef --- /dev/null +++ b/apps/website/src/components/queenRoadmapScene.ts @@ -0,0 +1,711 @@ +// The ROADMAP comb in Babylon.js, drawn the way the Queen's field is (the user, +// 2026-09-27: "сделай также на поле как Babylon.js"). The same picture as the +// flat comb - an inverted pyramid of hex cells, apex down, the seed t27c at the +// point - but as the field draws a hive: a wall of hex prisms hanging in front +// of the player, tone-mapped and bloomed like QueenCombBabylon. +// +// * A cell's HEIGHT is how far its file has come: a built cell is a full +// honey prism, a cell in review stands half as tall, one a bee is building +// is a lit scaffold, a held one a low red slab, a free target a thin plate +// in its language's colour. Cells not filed yet are rims only. +// * SHIPS hover in front of the wall over the cells the Queen's board says are +// running, each with a beam into its cell. Nothing is drawn that the board +// did not report: the scene gets exactly the cells the flat comb gets. +// * THE QUEEN'S ROUND is a band of light climbing the wall from the apex, on +// the board's own pulse (pulsePhase, the same function the flat comb uses). +// * The pointer lifts the cell under it toward the hand, as on the field; a +// click opens the cell's issue. Dragging tilts the wall a little; the wheel +// is left to the page, which scrolls. +// +// Renders ON DEMAND like the research city: while nothing moves, one frame per +// change or camera move. With motion allowed and something to animate (ships, +// the round, a raid) it runs at 30 FPS, and only while the canvas is on screen. +// DPR capped at 1.5. +import { Engine } from '@babylonjs/core/Engines/engine' +import { Scene } from '@babylonjs/core/scene' +import { ArcRotateCamera } from '@babylonjs/core/Cameras/arcRotateCamera' +import { Vector3, Matrix, Quaternion } from '@babylonjs/core/Maths/math.vector' +import { Color3, Color4 } from '@babylonjs/core/Maths/math.color' +import { Mesh } from '@babylonjs/core/Meshes/mesh' +import { VertexData } from '@babylonjs/core/Meshes/mesh.vertexData' +import { VertexBuffer } from '@babylonjs/core/Buffers/buffer' +import { TransformNode } from '@babylonjs/core/Meshes/transformNode' +import { StandardMaterial } from '@babylonjs/core/Materials/standardMaterial' +import { DynamicTexture } from '@babylonjs/core/Materials/Textures/dynamicTexture' +import { ImageProcessingConfiguration } from '@babylonjs/core/Materials/imageProcessingConfiguration' +import { CreateCylinder } from '@babylonjs/core/Meshes/Builders/cylinderBuilder' +import { CreateSphere } from '@babylonjs/core/Meshes/Builders/sphereBuilder' +import { CreateBox } from '@babylonjs/core/Meshes/Builders/boxBuilder' +import { CreatePlane } from '@babylonjs/core/Meshes/Builders/planeBuilder' +import { CreateLineSystem } from '@babylonjs/core/Meshes/Builders/linesBuilder' +import { HemisphericLight } from '@babylonjs/core/Lights/hemisphericLight' +import { DirectionalLight } from '@babylonjs/core/Lights/directionalLight' +import { GlowLayer } from '@babylonjs/core/Layers/glowLayer' +import '@babylonjs/core/Layers/effectLayerSceneComponent' +import '@babylonjs/core/Meshes/thinInstanceMesh' +import { Ray } from '@babylonjs/core/Culling/ray' +import { PointerEventTypes } from '@babylonjs/core/Events/pointerEvents' +import type { LinesMesh } from '@babylonjs/core/Meshes/linesMesh' +import type { Material } from '@babylonjs/core/Materials/material' +import { pulsePhase } from '../lib/roadmapGame' + +export type RoadmapCellKind = 'seed' | 'built' | 'cracked' | 'review' | 'building' | 'held' | 'target' | 'future' + +export interface RoadmapSceneCell { + /** Centre in the flat comb's coordinates (SVG units, y grows downward). */ + x: number + y: number + kind: RoadmapCellKind + /** The issue the cell is; null for the seed and for cells not filed yet. */ + number: number | null + /** The colour of the language the cell's file is written in. */ + langHex: string + /** In today's raid sector. */ + raid: boolean +} + +export interface RoadmapSceneInput { + cells: RoadmapSceneCell[] + /** The flat comb's hex radius and frame, so both drawings share one geometry. */ + s: number + frame: { cx: number; top: number; apex: number; halfTop: number } + pulse: { lastRoundAt: string | null; roundSeconds: number } | null + motion: 'static' | 'interactive' +} + +export interface RoadmapSceneEvents { + /** The cell under the pointer changed; null when it left every cell. */ + onHover(index: number | null, x: number, y: number): void + /** A click or tap on a cell, with the pointer type that made it. */ + onPick(index: number, pointerType: string, x: number, y: number): void + /** The WebGL context is gone; the page falls back to the flat comb. */ + onLost(message: string): void +} + +export interface RoadmapSceneHandle { + update(input: RoadmapSceneInput): void + dispose(): void +} + +export const DPR_CAP = 1.5 +export const SCENE_FPS = 30 + +const MARK = '#08fab5' +const GOLD = '#ffd35a' +const RAID = '#ffae3a' +const HELD = '#ff8080' +const CRACK_RIM = '#ff6b6b' + +/** How far a cell stands out of the wall, by state: how far its file has come. */ +const HEIGHT: Record = { + seed: 2.6, + built: 1.7, + cracked: 1.6, + review: 1.0, + building: 0.8, + held: 0.35, + target: 0.14, + future: 0, +} +const RIM: Record, [string, number]> = { + seed: ['#eafff7', 1], + built: [MARK, 0.95], + cracked: [CRACK_RIM, 1], + review: [GOLD, 0.95], + building: [MARK, 1], + held: [HELD, 0.9], + future: [MARK, 0.14], +} +/** Cell radius in wall units: the flat comb's s - 1.5 over s. */ +const R = 0.9 +const LIFT = 0.5 +const TILT = { alpha: -Math.PI / 2 + 0.3, beta: Math.PI / 2 - 0.2 } +const FOV = 0.72 + +function hexCorners(x: number, y: number, r: number, z: number): Vector3[] { + const out: Vector3[] = [] + for (let k = 0; k <= 6; k += 1) { + const a = (Math.PI / 180) * (90 - 60 * k) + out.push(new Vector3(x + r * Math.cos(a), y + r * Math.sin(a), z)) + } + return out +} + +export function mountRoadmapScene( + canvas: HTMLCanvasElement, + initial: RoadmapSceneInput, + events: RoadmapSceneEvents, +): RoadmapSceneHandle { + // Throws where WebGL is unavailable; the caller draws the flat comb instead. + const engine = new Engine(canvas, true, { preserveDrawingBuffer: false, stencil: false, doNotHandleTouchAction: true }, false) + engine.setHardwareScalingLevel(1 / Math.min(DPR_CAP, window.devicePixelRatio || 1)) + const scene = new Scene(engine) + scene.clearColor = new Color4(0, 0, 0, 0) + scene.skipPointerMovePicking = true + // The field's look: ACES tone mapping with a little contrast, and a bloom on + // every emissive rim, ship and beam. + const ipc = scene.imageProcessingConfiguration + ipc.toneMappingEnabled = true + ipc.toneMappingType = ImageProcessingConfiguration.TONEMAPPING_ACES + ipc.contrast = 1.22 + ipc.exposure = 1.08 + const glow = new GlowLayer('rm-glow', scene, { blurKernelSize: 16 }) + glow.intensity = 0.55 + + const camera = new ArcRotateCamera('rm-cam', TILT.alpha, TILT.beta, 30, new Vector3(0, 0, -0.5), scene) + camera.fov = FOV + camera.minZ = 0.1 + camera.maxZ = 500 + camera.panningSensibility = 0 + camera.lowerAlphaLimit = -Math.PI / 2 - 0.55 + camera.upperAlphaLimit = -Math.PI / 2 + 0.55 + camera.lowerBetaLimit = Math.PI / 2 - 0.45 + camera.upperBetaLimit = Math.PI / 2 + 0.3 + // The page scrolls under the wheel; the wall only tilts under a drag, and only + // with a mouse or a pen: on a touch screen a drag must scroll the page. + camera.inputs.removeByType('ArcRotateCameraMouseWheelInput') + camera.inputs.removeByType('ArcRotateCameraKeyboardMoveInput') + const finePointer = window.matchMedia('(pointer: fine)').matches + if (finePointer) camera.attachControl(canvas, true) + + const sky = new HemisphericLight('rm-sky', new Vector3(0, 1, -0.6), scene) + sky.intensity = 0.42 + sky.groundColor = Color3.FromHexString('#02140d') + // From the upper left and in front: the faces take the light, the sides of + // each prism a shade apart, which is what reads as depth. + const key = new DirectionalLight('rm-key', new Vector3(0.7, -0.45, 0.55), scene) + key.intensity = 1.15 + key.diffuse = Color3.FromHexString('#fff0bf') + + // ---- rendering on demand -------------------------------------------------- + let disposed = false + let dirty = true + let raf = 0 + let lastFrame = 0 + let onScreen = true + let staticTimer = 0 + let animate: ((t: number) => void) | null = null + // Something on the wall moves: a ship, the round, a raid rim or a scaffold. + let motionful = false + let lastInput: RoadmapSceneInput = initial + const moving = () => lastInput.motion === 'interactive' && motionful + const frame = (t: number) => { + raf = 0 + if (disposed) return + if (moving() && onScreen && !document.hidden) { + if (t - lastFrame >= 1000 / SCENE_FPS - 1) { + lastFrame = t + animate?.(t) + scene.render() + dirty = false + } + raf = window.requestAnimationFrame(frame) + return + } + if (dirty) { + animate?.(t) + scene.render() + dirty = false + } + } + const requestRender = () => { + dirty = true + if (!raf && !disposed) raf = window.requestAnimationFrame(frame) + } + camera.onViewMatrixChangedObservable.add(requestRender) + const visibility = new IntersectionObserver((entries) => { + onScreen = entries.some((e) => e.isIntersecting) + if (onScreen) requestRender() + }) + visibility.observe(canvas) + const onVisibility = () => { if (!document.hidden) requestRender() } + document.addEventListener('visibilitychange', onVisibility) + engine.onContextLostObservable.add(() => events.onLost('WebGL context lost')) + + // ---- geometry shared by every build --------------------------------------- + const prismBase = (name: string) => { + const m = CreateCylinder(name, { tessellation: 6, diameter: 2 * R, height: 1 }, scene) + // Axis toward the camera (-z): the cap is the cell's face. + m.rotation.x = -Math.PI / 2 + m.bakeCurrentTransformIntoVertices() + // Pointy-top, like the flat comb: taller than wide. + const bb = m.getBoundingInfo().boundingBox + if (bb.maximum.x - bb.minimum.x > bb.maximum.y - bb.minimum.y) { + m.rotation.z = Math.PI / 6 + m.bakeCurrentTransformIntoVertices() + } + return m + } + + let meshes: Array = [] + let materials: Material[] = [] + let textures: DynamicTexture[] = [] + const keep = (m: T): T => { meshes.push(m); return m } + const mat = (m: T): T => { materials.push(m); return m } + const clear = () => { + // Each node is in the list, so none is disposed through its parent. + for (const m of meshes) m.dispose(true, false) + for (const m of materials) m.dispose() + for (const t of textures) t.dispose() + meshes = [] + materials = [] + textures = [] + animate = null + motionful = false + } + const solid = (name: string, hex: string, emissive: number, alpha = 1) => { + const m = mat(new StandardMaterial(name, scene)) + m.diffuseColor = Color3.FromHexString(hex) + m.emissiveColor = Color3.FromHexString(hex).scale(emissive) + m.specularColor = new Color3(0.45, 0.42, 0.3) + m.specularPower = 40 + m.alpha = alpha + return m + } + const light = (name: string, hex: string, alpha = 1) => { + const m = mat(new StandardMaterial(name, scene)) + m.diffuseColor = Color3.Black() + m.specularColor = Color3.Black() + m.emissiveColor = Color3.FromHexString(hex) + m.disableLighting = true + m.alpha = alpha + m.backFaceCulling = false + return m + } + + // ---- state the pointer and the animation read ----------------------------- + let world: Array<{ X: number; Y: number; h: number }> = [] + let slots: Array<{ mesh: Mesh; i: number } | null> = [] + let hoverRing: LinesMesh | null = null + let lifted = -1 + let pulseBand: Mesh | null = null + let toWorldY = (y: number) => y + let frameGeom = { top: 0, apex: 0, halfTop: 0, s: 1 } + + const placeInstance = (index: number, up: number) => { + const slot = slots[index] + const w = world[index] + if (!slot || !w) return + const m = Matrix.Scaling(1, 1, Math.max(w.h, 0.02)).multiply(Matrix.Translation(w.X, w.Y, -Math.max(w.h, 0.02) / 2 - up)) + slot.mesh.thinInstanceSetMatrixAt(slot.i, m, true) + } + + const placePulse = () => { + if (!pulseBand) return + const p = lastInput.pulse + const phase = p ? pulsePhase(p.lastRoundAt, p.roundSeconds, Date.now()) : null + if (phase === null) { + pulseBand.isVisible = false + return + } + pulseBand.isVisible = true + const { top, apex, halfTop, s } = frameGeom + const centre = apex - phase * (apex - top) + const lo = Math.min(apex, centre + 1.2 * s) + const hi = Math.max(top, centre - 1.2 * s) + const half = (y: number) => (halfTop * (apex - y)) / (apex - top) / s + const z = -2.8 + const positions = [ + -half(lo), toWorldY(lo), z, + half(lo), toWorldY(lo), z, + half(hi), toWorldY(hi), z, + -half(hi), toWorldY(hi), z, + ] + pulseBand.updateVerticesData(VertexBuffer.PositionKind, positions) + pulseBand.refreshBoundingInfo() + } + + const fit = () => { + const w = canvas.clientWidth + const h = canvas.clientHeight + if (!w || !h) return + const { top, apex, halfTop, s } = frameGeom + const halfH = (apex - top) / (2 * s) + 0.8 + const halfW = halfTop / s + 0.8 + const t = Math.tan(FOV / 2) + let r = Math.max(halfH / t, halfW / (t * (w / h))) + 1.2 + // The wall is tilted, so the near corner grows: step back until every + // corner of the frame, at the depth of the tallest cells too, is in view. + const corners: Vector3[] = [] + for (const z of [0, -HEIGHT.seed]) { + corners.push(new Vector3(-halfTop / s, toWorldY(top), z), new Vector3(halfTop / s, toWorldY(top), z), new Vector3(0, toWorldY(apex), z)) + } + camera.upperRadiusLimit = null + camera.lowerRadiusLimit = null + for (let i = 0; i < 24; i += 1) { + camera.radius = r + const vp = camera.getViewMatrix(true).multiply(camera.getProjectionMatrix(true)) + const inside = corners.every((c) => { + const q = Vector3.TransformCoordinates(c, vp) + return Math.abs(q.x) <= 0.94 && Math.abs(q.y) <= 0.94 + }) + if (inside) break + r *= 1.06 + } + camera.radius = r + camera.lowerRadiusLimit = r * 0.7 + camera.upperRadiusLimit = r * 1.15 + } + + const build = (input: RoadmapSceneInput) => { + clear() + const { cells, s, frame: f } = input + frameGeom = { top: f.top, apex: f.apex, halfTop: f.halfTop, s } + const mid = (f.top + f.apex) / 2 + toWorldY = (y: number) => (mid - y) / s + world = cells.map((c) => ({ X: (c.x - f.cx) / s, Y: (mid - c.y) / s, h: HEIGHT[c.kind] })) + slots = cells.map(() => null) + + // The frame of the mark and a faint plate behind the comb. + const tl = new Vector3(-f.halfTop / s, toWorldY(f.top), 0.04) + const tr = new Vector3(f.halfTop / s, toWorldY(f.top), 0.04) + const ap = new Vector3(0, toWorldY(f.apex), 0.04) + const frameLines = keep(CreateLineSystem('rm-frame', { lines: [[tl, tr, ap, tl]] }, scene)) + frameLines.color = Color3.FromHexString(MARK) + frameLines.alpha = 0.4 + frameLines.isPickable = false + glow.referenceMeshToUseItsOwnMaterial(frameLines) + const plate = keep(new Mesh('rm-plate', scene)) + const pd = new VertexData() + pd.positions = [tl.x, tl.y, 0.06, tr.x, tr.y, 0.06, ap.x, ap.y, 0.06] + pd.indices = [0, 1, 2] + pd.applyToMesh(plate) + plate.material = light('rm-plate-mat', MARK, 0.045) + plate.isPickable = false + glow.addExcludedMesh(plate) + + // Prisms: one thin-instanced mesh per state (per language for the targets), + // not a mesh per cell. + const batches = new Map() + cells.forEach((c, i) => { + if (c.kind === 'future' || c.kind === 'seed') return + const k = c.kind === 'target' ? `target:${c.langHex}` : c.kind + const b = batches.get(k) ?? { hex: c.langHex, kind: c.kind, items: [] } + b.items.push(i) + batches.set(k, b) + }) + let buildingMat: StandardMaterial | null = null + for (const [k, b] of batches) { + const m = keep(prismBase(`rm-prism-${k}`)) + m.isPickable = false + m.alwaysSelectAsActiveMesh = true + let material: StandardMaterial + switch (b.kind) { + case 'built': material = solid('rm-built', '#e89b12', 0.12); break + case 'cracked': material = solid('rm-cracked', '#8a6a3a', 0.08); break + case 'review': material = solid('rm-review', GOLD, 0.22, 0.45); break + case 'building': material = solid('rm-building', MARK, 0.45, 0.5); buildingMat = material; break + case 'held': material = solid('rm-held', HELD, 0.22, 0.4); break + default: material = solid(`rm-target-${b.hex}`, b.hex, 0.35, 0.34) + } + m.material = material + // Broad fills must not bloom into their neighbours; the rims carry the glow. + if (b.kind !== 'building') glow.addExcludedMesh(m) + const matrices = new Float32Array(16 * b.items.length) + b.items.forEach((cellIndex, j) => { + const w = world[cellIndex] + Matrix.Scaling(1, 1, w.h).multiply(Matrix.Translation(w.X, w.Y, -w.h / 2)).copyToArray(matrices, 16 * j) + slots[cellIndex] = { mesh: m, i: j } + }) + m.thinInstanceSetBuffer('matrix', matrices, 16, false) + } + + // The seed: the one hand-written thing, the tallest cell, lit, named. + const seedIndex = cells.findIndex((c) => c.kind === 'seed') + if (seedIndex >= 0) { + const w = world[seedIndex] + const seed = keep(prismBase('rm-seed')) + seed.scaling.z = w.h + seed.position.set(w.X, w.Y, -w.h / 2) + seed.material = solid('rm-seed-mat', MARK, 0.85) + seed.isPickable = false + const dt = new DynamicTexture('rm-seed-label', { width: 256, height: 128 }, scene, true) + textures.push(dt) + dt.hasAlpha = true + const ctx = dt.getContext() as unknown as CanvasRenderingContext2D + ctx.clearRect(0, 0, 256, 128) + dt.drawText('t27c', null, 86, 'bold 72px system-ui, sans-serif', '#02140d', 'transparent', true, true) + const label = keep(CreatePlane('rm-seed-text', { width: 1.3, height: 0.65 }, scene)) + label.position.set(w.X, w.Y, -w.h - 0.02) + const lm = mat(new StandardMaterial('rm-seed-text-mat', scene)) + lm.diffuseTexture = dt + lm.useAlphaFromDiffuseTexture = true + lm.emissiveColor = Color3.White() + lm.disableLighting = true + label.material = lm + label.isPickable = false + glow.addExcludedMesh(label) + } + + // Rims on each cell's face; today's raid targets get their own, pulsing. + const lines: Vector3[][] = [] + const colours: Color4[][] = [] + const raidLines: Vector3[][] = [] + const crackLines: Vector3[][] = [] + cells.forEach((c, i) => { + const w = world[i] + const corners = hexCorners(w.X, w.Y, R, -w.h - 0.01) + if (c.kind === 'target' && c.raid) { + raidLines.push(corners) + return + } + const [hex, alpha] = c.kind === 'target' ? [c.langHex, 0.95] : RIM[c.kind] + const col = Color3.FromHexString(hex) + lines.push(corners) + colours.push(corners.map(() => new Color4(col.r, col.g, col.b, alpha))) + if (c.kind === 'cracked') { + const z = -w.h - 0.015 + const x0 = w.X - 0.55 + const y0 = w.Y + 0.35 + crackLines.push([ + new Vector3(x0, y0, z), + new Vector3(x0 + 0.35, y0 - 0.3, z), + new Vector3(x0 + 0.5, y0 - 0.05, z), + new Vector3(x0 + 0.85, y0 - 0.6, z), + new Vector3(x0 + 1.1, y0 - 0.4, z), + ]) + } + }) + if (lines.length) { + const rims = keep(CreateLineSystem('rm-rims', { lines, colors: colours, useVertexAlpha: true }, scene)) + rims.isPickable = false + glow.referenceMeshToUseItsOwnMaterial(rims) + } + let raidRims: LinesMesh | null = null + if (raidLines.length) { + raidRims = keep(CreateLineSystem('rm-raid', { lines: raidLines }, scene)) + raidRims.color = Color3.FromHexString(RAID) + raidRims.isPickable = false + glow.referenceMeshToUseItsOwnMaterial(raidRims) + } + if (crackLines.length) { + const cracks = keep(CreateLineSystem('rm-cracks', { lines: crackLines }, scene)) + cracks.color = Color3.FromHexString('#2a0a0a') + cracks.isPickable = false + glow.addExcludedMesh(cracks) + } + hoverRing = keep(CreateLineSystem('rm-hover', { lines: [hexCorners(0, 0, R * 1.04, 0)] }, scene)) + hoverRing.color = Color3.FromHexString('#eafff7') + hoverRing.isPickable = false + hoverRing.isVisible = false + glow.referenceMeshToUseItsOwnMaterial(hoverRing) + lifted = -1 + + // The Queen's round: a band of light across the wall, in front of it. + pulseBand = keep(new Mesh('rm-pulse', scene)) + const bd = new VertexData() + bd.positions = new Array(12).fill(0) + bd.indices = [0, 1, 2, 0, 2, 3] + bd.applyToMesh(pulseBand, true) + pulseBand.material = light('rm-pulse-mat', MARK, 0.13) + pulseBand.isPickable = false + pulseBand.alwaysSelectAsActiveMesh = true + glow.addExcludedMesh(pulseBand) + + // Ships: one per cell the board says a bee is building, with a beam into it. + const hullMat = solid('rm-hull', '#dffcf3', 0.12) + const cockpitMat = light('rm-cockpit', MARK) + const flameMat = light('rm-flame', RAID) + const beamMat = light('rm-beam', MARK, 0.22) + const ships: Array<{ node: TransformNode; flame: Mesh; beam: Mesh; baseY: number; face: Vector3; seed: number }> = [] + cells.forEach((c, i) => { + if (c.kind !== 'building') return + const w = world[i] + const n = ships.length + const side = n % 2 === 0 ? -1 : 1 + const node = keep(new TransformNode(`rm-ship-${i}`, scene)) + node.position.set(w.X + side * 1.1, w.Y + 2.6, -4.2) + node.rotation.z = side * 0.28 + node.scaling.setAll(1.5) + const hull = keep(CreateCylinder(`rm-hull-${i}`, { tessellation: 3, diameterTop: 0, diameterBottom: 0.95, height: 1.35 }, scene)) + hull.rotation.z = Math.PI + hull.scaling.z = 0.45 + hull.parent = node + hull.material = hullMat + const wings = keep(CreateBox(`rm-wings-${i}`, { width: 1.7, height: 0.1, depth: 0.42 }, scene)) + wings.position.y = 0.3 + wings.parent = node + wings.material = hullMat + const cockpit = keep(CreateSphere(`rm-cockpit-${i}`, { diameter: 0.34, segments: 8 }, scene)) + cockpit.position.set(0, 0.05, -0.2) + cockpit.parent = node + cockpit.material = cockpitMat + const flame = keep(CreateCylinder(`rm-flame-${i}`, { tessellation: 8, diameterTop: 0, diameterBottom: 0.36, height: 0.6 }, scene)) + flame.position.y = 1.0 + flame.parent = node + flame.material = flameMat + const beam = keep(CreateCylinder(`rm-beam-${i}`, { tessellation: 20, diameterTop: 0.12, diameterBottom: 1.45, height: 1, cap: Mesh.NO_CAP }, scene)) + beam.material = beamMat + beam.rotationQuaternion = new Quaternion() + for (const m of [hull, wings, cockpit, flame, beam]) m.isPickable = false + ships.push({ node, flame, beam, baseY: node.position.y, face: new Vector3(w.X, w.Y, -w.h), seed: n }) + }) + const up = Vector3.Up() + const dir = new Vector3() + const nose = new Vector3() + const aimBeam = (ship: (typeof ships)[number]) => { + // The nose of a ship tilted by rotation.z, in wall units. + const tilt = ship.node.rotation.z + nose.set(ship.node.position.x + Math.sin(tilt) * 1.02, ship.node.position.y - Math.cos(tilt) * 1.02, ship.node.position.z) + nose.subtractToRef(ship.face, dir) + const length = dir.length() + dir.scaleInPlace(1 / length) + Quaternion.FromUnitVectorsToRef(up, dir, ship.beam.rotationQuaternion!) + ship.beam.scaling.y = length + ship.beam.position.set((nose.x + ship.face.x) / 2, (nose.y + ship.face.y) / 2, (nose.z + ship.face.z) / 2) + } + ships.forEach(aimBeam) + + canvas.dataset.scene = 'babylon' + canvas.dataset.cells = String(cells.filter((c) => c.number !== null).length) + canvas.dataset.ships = String(ships.length) + canvas.dataset.raid = String(raidLines.length) + + motionful = ships.length > 0 || raidLines.length > 0 || buildingMat !== null || input.pulse !== null + animate = (t: number) => { + placePulse() + if (lastInput.motion !== 'interactive') return + const sec = t / 1000 + for (const ship of ships) { + ship.node.position.y = ship.baseY + Math.sin(sec * 2.4 + ship.seed) * 0.15 + ship.flame.scaling.y = 0.8 + 0.35 * Math.abs(Math.sin(sec * 17 + ship.seed)) + aimBeam(ship) + } + beamMat.alpha = 0.14 + 0.1 * (Math.sin(sec * 4.5) + 1) / 2 + if (buildingMat) buildingMat.emissiveColor = Color3.FromHexString(MARK).scale(0.3 + 0.35 * (Math.sin(sec * 3) + 1) / 2) + if (raidRims) raidRims.alpha = 0.45 + 0.55 * (Math.sin(sec * 3.9) + 1) / 2 + } + fit() + } + + // ---- the pointer: lift, card, click ---------------------------------------- + const pickRay = new Ray(Vector3.Zero(), Vector3.Forward(), 1e6) + const cellAt = (px: number, py: number): number => { + scene.createPickingRayToRef(px, py, null, pickRay, camera) + const dz = pickRay.direction.z + if (Math.abs(dz) < 1e-6) return -1 + let best = -1 + let bestD = R * R + // Solve against the plane of each candidate's own face, so a tall prism is + // found where it is drawn, not where the wall is. + lastInput.cells.forEach((c, i) => { + if (c.number === null && c.kind !== 'seed') return + const w = world[i] + const t = (-w.h - pickRay.origin.z) / dz + if (t < 0) return + const dx = pickRay.origin.x + pickRay.direction.x * t - w.X + const dy = pickRay.origin.y + pickRay.direction.y * t - w.Y + const d = dx * dx + dy * dy + if (d < bestD) { + bestD = d + best = i + } + }) + return best + } + const setLift = (index: number) => { + if (index === lifted) return + if (lifted >= 0) placeInstance(lifted, 0) + lifted = index + if (index >= 0) placeInstance(index, LIFT) + if (hoverRing) { + if (index >= 0) { + const w = world[index] + const up = slots[index] ? LIFT : 0 + hoverRing.position.set(w.X, w.Y, -w.h - up - 0.03) + hoverRing.isVisible = true + } else { + hoverRing.isVisible = false + } + } + requestRender() + } + let downAt: [number, number] | null = null + let travelled = 0 + scene.onPointerObservable.add((info) => { + const x = scene.pointerX + const y = scene.pointerY + if (info.type === PointerEventTypes.POINTERMOVE) { + if (downAt) { + travelled += Math.abs(x - downAt[0]) + Math.abs(y - downAt[1]) + downAt = [x, y] + } + const index = cellAt(x, y) + canvas.style.cursor = index >= 0 ? 'pointer' : finePointer ? 'grab' : 'default' + if (index !== lifted) { + setLift(index) + events.onHover(index >= 0 ? index : null, x, y) + } + } else if (info.type === PointerEventTypes.POINTERDOWN) { + downAt = [x, y] + travelled = 0 + } else if (info.type === PointerEventTypes.POINTERUP) { + if (downAt && travelled <= 6) { + const index = cellAt(x, y) + if (index >= 0) { + setLift(index) + events.onPick(index, (info.event as PointerEvent).pointerType ?? 'mouse', x, y) + } + } + downAt = null + } + }) + const onLeave = () => { + setLift(-1) + events.onHover(null, 0, 0) + } + canvas.addEventListener('pointerleave', onLeave) + + const ro = new ResizeObserver(() => { + engine.resize() + fit() + requestRender() + }) + ro.observe(canvas) + + const armStatic = () => { + window.clearInterval(staticTimer) + staticTimer = 0 + // Without motion the round still moves, once a second, as on the flat comb. + if (lastInput.motion === 'static' && lastInput.pulse) staticTimer = window.setInterval(requestRender, 1000) + } + + const update = (input: RoadmapSceneInput) => { + if (disposed) return + const rebuild = + input.cells !== lastInput.cells || + input.s !== lastInput.s || + input.frame.top !== lastInput.frame.top || + input.frame.apex !== lastInput.frame.apex || + input.frame.halfTop !== lastInput.frame.halfTop || + (input.pulse === null) !== (lastInput.pulse === null) + lastInput = input + if (rebuild) build(input) + camera.inertia = input.motion === 'interactive' ? 0.9 : 0 + armStatic() + requestRender() + } + + build(initial) + camera.inertia = initial.motion === 'interactive' ? 0.9 : 0 + armStatic() + scene.onAfterRenderObservable.addOnce(() => { canvas.dataset.ready = '1' }) + requestRender() + + return { + update, + dispose() { + disposed = true + window.clearInterval(staticTimer) + if (raf) window.cancelAnimationFrame(raf) + ro.disconnect() + visibility.disconnect() + document.removeEventListener('visibilitychange', onVisibility) + canvas.removeEventListener('pointerleave', onLeave) + clear() + glow.dispose() + scene.dispose() + engine.dispose() + }, + } +} diff --git a/apps/website/src/data/blog/bodies/a-small-agent-needs-an-exact-judge.ts b/apps/website/src/data/blog/bodies/a-small-agent-needs-an-exact-judge.ts index 08536c9c67..87eecde8e8 100644 --- a/apps/website/src/data/blog/bodies/a-small-agent-needs-an-exact-judge.ts +++ b/apps/website/src/data/blog/bodies/a-small-agent-needs-an-exact-judge.ts @@ -112,7 +112,7 @@ export const body: Block[] = [ { kind: 'ul', items: [ - "No IGLA model has run on a board. Every throughput figure here is a ceiling derived from memory bandwidth and LUT counts.", + "Update 2026-09-27: the trained tern_tc model's 320-input weight matrices ran bit-exact on an AX7203 (see the follow-up post). No forward pass has run on a board, and every throughput figure here is still a ceiling derived from memory bandwidth and LUT counts.", "The code ability of a 13M model on .t27 is unknown. The HumanEval figures above are for Python and for another model family.", "The open DDR3 path on Artix-7 has passed memtest on hardware only since nextpnr-xilinx 0.9.5 (13 September 2026). The independent UberDDR3 test reached a 333 MHz DDR clock, not the 400 MHz the AX7203 is rated for.", "Integer inference must be shown to cost little quality against the float model before receipts can rest on it.", @@ -234,7 +234,7 @@ export const ruBody: Block[] = [ { kind: 'ul', items: [ - "Ни одна модель IGLA ещё не запускалась на плате. Каждая цифра скорости здесь — потолок, выведенный из пропускной способности памяти и числа LUT.", + "Обновление 2026-09-27: матрицы весов с входом 320 обученной модели tern_tc посчитались на AX7203 бит-точно (см. следующий пост). Прямой проход на плате не запускался, и каждая цифра скорости здесь по-прежнему потолок, выведенный из пропускной способности памяти и числа LUT.", "Способность модели на 13M писать .t27 неизвестна. Цифры HumanEval выше относятся к Python и к другому семейству моделей.", "Открытый путь к DDR3 на Artix-7 проходит memtest на железе только с nextpnr-xilinx 0.9.5 (13 сентября 2026). Независимый тест UberDDR3 достиг частоты DDR 333 МГц, а не 400 МГц, на которые рассчитана AX7203.", "Нужно показать, что целочисленный вывод почти не теряет в качестве по сравнению с плавающей точкой. Только после этого на него могут опираться квитанции.", diff --git a/apps/website/src/data/blog/bodies/golden-ratio-weights-ran-on-the-board.ts b/apps/website/src/data/blog/bodies/golden-ratio-weights-ran-on-the-board.ts new file mode 100644 index 0000000000..4d9763894c --- /dev/null +++ b/apps/website/src/data/blog/bodies/golden-ratio-weights-ran-on-the-board.ts @@ -0,0 +1,175 @@ +import type { Block } from '../types' + +export const body: Block[] = [ + { + kind: 'p', + text: "The TNF paper's weight format is GFTernary: a weight is t*phi, with t in {-1, 0, +1}. Its activations live in Z[phi], the numbers a + b*phi with integer a and b. The paper claims that a layer's linear path is exact in Z[phi] and needs no multiplier. The AX7203 node from the previous post computes signed 32-trit ternary dot products and signs every answer. We asked whether that node, unchanged, can compute a GFTernary layer on Z[phi] activations exactly.", + }, + { + kind: 'h', + text: 'Why no multiplier is needed', + }, + { + kind: 'p', + text: 'Because phi^2 = phi + 1, applying a weight is the Fibonacci step: t*phi*(a + b*phi) = t*(b + (a + b)*phi). Summed over a row, one output is W.b + (W.a + W.b)*phi. W.a and W.b are ternary-by-integer dot products. With a and b in [-127, 127], each one is six balanced-ternary digit planes, which makes 12 node jobs per 32-wide chunk. The phi itself costs one integer add per output, done on the host.', + }, + { + kind: 'p', + text: "This is the same point as an earlier post, 'The golden ratio in this format is a scale factor, not information', now measured on silicon. In the weights, phi is a fixed linear map on Z[phi], not a product the hardware has to form. The node's work in this run is exactly the kind of work it did for tern_tc: ternary dots, nothing else.", + }, + { + kind: 'h', + text: 'Registered before it ran', + }, + { + kind: 'p', + text: 'The command, the input, the expected numbers and the rules were committed at 03:32:06 UTC, and the run started at 03:32:44. The input was layer 0 of the trained tern_tc model: all 7 ternary matrices, 2,816 outputs, one Z[phi] activation vector from a fixed seed, and 24 jobs in flight. The rules: the run happens once, and its result is recorded whatever it is. A UART slip would count against the link, not against the Z[phi] claim, and would still be recorded as a fail.', + }, + { + kind: 'table', + head: ['', 'Registered', 'Board'], + rows: [ + ['Jobs', '403,200', '403,200 of 403,200 sent'], + ['Receipts verified', '403,200 / 403,200', '403,200 / 403,200'], + ['Z[phi] rows bit-exact', '5,632 / 5,632', '5,632 / 5,632'], + ['Rejected', 'none', 'none'], + ['Time', 'about 86 s', '85.35 s, 4,724 answers/s'], + ], + }, + { + kind: 'p', + text: 'Every answer passed the same five checks as in the tern_tc run: status, nonce, node id, a SipHash tag recomputed under the key, and y. The oracle is plain Z[phi] multiplication, (a, b)(c, d) = (ac + bd, ad + bc + bd), applied to each weight and activation pair and summed. It uses neither the Fibonacci shortcut nor the digit split. Each output is checked as two rows, the rational part and the phi part minus the rational part. Both are bit-exact exactly when the output assembled from the board\'s answers equals the oracle in Z[phi].', + }, + { + kind: 'p', + text: 'Before the board run, the self-test showed each check able to fail:', + }, + { + kind: 'table', + head: ['Wrong on purpose', 'Result'], + rows: [ + ['an oracle that uses phi^2 = 1', '12 / 24 rows: every phi row fails'], + ['weights read as t instead of t*phi', '0 / 24 rows'], + ['the host drops the 3^0 digit plane', '0 / 24 rows'], + ['a validly signed wrong answer', "rejected as 'lie'"], + ['a cell signing with another key', '0 receipts'], + ], + }, + { + kind: 'p', + text: "The node's own Verilog, unchanged, also ran a request stream of the same kind in simulation, with synthetic weights for one matrix: 7,680 of 7,680 receipts and 128 of 128 rows.", + }, + { + kind: 'h', + text: 'What this is not', + }, + { + kind: 'ul', + items: [ + "Not the TNF accumulator. Nothing was rounded, and TNF16 has no RTL. The digit recombination and the phi step are host arithmetic, one add per output for the phi.", + 'Not the model on real data. The activations are one synthetic Z[phi] vector, and only layer 0 ran.', + 'Not fast. 12,902,400 ternary multiply-accumulates in 85.35 s is about 151,000 per second (derived). The UART sets the pace, and a laptop CPU does this in milliseconds.', + "Not a public proof. SipHash is a shared-key MAC: it tells the key holder which node answered, and it does not stop an operator forging their own receipts.", + ], + }, + { + kind: 'h', + text: 'Next', + }, + { + kind: 'ol', + items: [ + 'All six layers: 2,419,200 jobs, about 8.5 minutes at the measured rate (derived).', + 'Real activations: take the Z[phi] inputs a TNF forward pass would produce, not a seeded vector.', + 'The accumulator: TNF rounding in RTL, so the part of the claim that is not a dot product can be tested too.', + 'Receipts anyone can check: a Merkle root per run, and random re-execution or Freivalds checks.', + ], + }, +] + +export const ruBody: Block[] = [ + { + kind: 'p', + text: 'Формат весов из статьи о TNF — GFTernary: вес равен t*phi, где t из {-1, 0, +1}. Активации живут в Z[phi], это числа a + b*phi с целыми a и b. Статья утверждает, что линейный путь слоя точен в Z[phi] и не требует умножителя. Узел на AX7203 из прошлого поста считает знаковые тернарные скалярные произведения на 32 трита и подписывает каждый ответ. Мы проверили, может ли этот узел без изменений точно посчитать слой GFTernary на активациях из Z[phi].', + }, + { + kind: 'h', + text: 'Почему умножитель не нужен', + }, + { + kind: 'p', + text: 'Поскольку phi^2 = phi + 1, применение веса — это шаг Фибоначчи: t*phi*(a + b*phi) = t*(b + (a + b)*phi). В сумме по строке один выход равен W.b + (W.a + W.b)*phi. W.a и W.b — скалярные произведения тернарных весов на целые. При a и b в диапазоне [-127, 127] каждое раскладывается на шесть сбалансированно-троичных разрядов, это 12 задач узла на кусок шириной 32. Сам phi стоит одно целочисленное сложение на выход, на хосте.', + }, + { + kind: 'p', + text: 'Это та же мысль, что в посте «Золотое сечение в этом формате — масштаб, а не информация», только теперь измеренная на кремнии. В весах phi — фиксированное линейное отображение на Z[phi], а не произведение, которое должно вычислять железо. Работа узла в этом прогоне ровно того же рода, что и для tern_tc: тернарные скалярные произведения, и ничего больше.', + }, + { + kind: 'h', + text: 'Зарегистрировано до запуска', + }, + { + kind: 'p', + text: 'Команда, входные данные, ожидаемые числа и правила были закоммичены в 03:32:06 UTC, а прогон начался в 03:32:44. На вход пошёл слой 0 обученной модели tern_tc: все 7 тернарных матриц, 2 816 выходов, один вектор активаций из Z[phi] из фиксированного seed, 24 задачи в полёте. Правила: прогон делается один раз, и его результат записывается, каким бы он ни был. Сбой UART засчитывался бы каналу, а не утверждению о Z[phi], и всё равно записывался бы как провал.', + }, + { + kind: 'table', + head: ['', 'Зарегистрировано', 'Плата'], + rows: [ + ['Задач', '403 200', 'отправлено 403 200 из 403 200'], + ['Квитанций проверено', '403 200 / 403 200', '403 200 / 403 200'], + ['Строк Z[phi] бит-точно', '5 632 / 5 632', '5 632 / 5 632'], + ['Отвергнуто', 'ничего', 'ничего'], + ['Время', 'около 86 с', '85,35 с, 4 724 ответа/с'], + ], + }, + { + kind: 'p', + text: 'Каждый ответ прошёл те же пять проверок, что в прогоне tern_tc: статус, nonce, id узла, тег SipHash, пересчитанный под ключом, и y. Оракул — обычное умножение в Z[phi], (a, b)(c, d) = (ac + bd, ad + bc + bd), применённое к каждой паре веса и активации и просуммированное. Он не использует ни шаг Фибоначчи, ни разложение на разряды. Каждый выход проверяется как две строки: рациональная часть и часть при phi минус рациональная. Обе бит-точны ровно тогда, когда выход, собранный из ответов платы, равен оракулу в Z[phi].', + }, + { + kind: 'p', + text: 'До прогона на плате самопроверка показала, что каждая проверка умеет падать:', + }, + { + kind: 'table', + head: ['Намеренная ошибка', 'Результат'], + rows: [ + ['оракул с phi^2 = 1', '12 / 24 строк: все строки при phi падают'], + ['веса прочитаны как t вместо t*phi', '0 / 24 строк'], + ['хост теряет разряд 3^0', '0 / 24 строк'], + ['неверный ответ с корректной подписью', "отвергнуто как 'lie'"], + ['ячейка подписывает чужим ключом', '0 квитанций'], + ], + }, + { + kind: 'p', + text: 'Собственный Verilog узла без изменений тоже прогнал в симуляции поток запросов того же вида, с синтетическими весами для одной матрицы: 7 680 из 7 680 квитанций и 128 из 128 строк.', + }, + { + kind: 'h', + text: 'Чем это не является', + }, + { + kind: 'ul', + items: [ + 'Это не аккумулятор TNF. Ничего не округлялось, и у TNF16 нет RTL. Сборка разрядов и шаг phi — арифметика хоста, одно сложение на выход для phi.', + 'Это не модель на реальных данных. Активации — один синтетический вектор из Z[phi], и прогнан только слой 0.', + 'Это не быстро. 12 902 400 тернарных умножений с накоплением за 85,35 с — примерно 151 000 в секунду (выведено). Темп задаёт UART, а процессор ноутбука делает это за миллисекунды.', + 'Это не публичное доказательство. SipHash — MAC с общим ключом: он сообщает владельцу ключа, какой узел ответил, и не мешает оператору подделать свои собственные квитанции.', + ], + }, + { + kind: 'h', + text: 'Дальше', + }, + { + kind: 'ol', + items: [ + 'Все шесть слоёв: 2 419 200 задач, около 8,5 минуты при измеренной скорости (выведено).', + 'Реальные активации: взять входы из Z[phi], которые дал бы прямой проход TNF, а не вектор из seed.', + 'Аккумулятор: округление TNF в RTL, чтобы проверить и ту часть утверждения, которая не сводится к скалярному произведению.', + 'Квитанции, которые может проверить любой: корень Меркла на прогон и случайное перевычисление или проверки Фрейвалдса.', + ], + }, +] diff --git a/apps/website/src/data/blog/bodies/trained-weights-ran-receipts-were-not-checked.ts b/apps/website/src/data/blog/bodies/trained-weights-ran-receipts-were-not-checked.ts new file mode 100644 index 0000000000..d76e25f0c2 --- /dev/null +++ b/apps/website/src/data/blog/bodies/trained-weights-ran-receipts-were-not-checked.ts @@ -0,0 +1,335 @@ +import type { Block } from '../types' + +export const body: Block[] = [ + { + kind: 'p', + text: "IGLA's board-sized model, tern_tc, has 9.08M parameters. 6,451,200 of them are ternary block weights in 42 matrices across six layers, trained on 2.0B tokens of code to 0.7613 validation bits per byte. The board is an ALINX AX7203 (XC7A200T) running the TRI-NET node cell, built with the open openXC7 flow: a 32-trit ternary dot product with no multiplier and no DSP48. It answers each job with the result, the job's nonce, a node id and a SipHash-2-4 tag over all of them. The host splits every 320-wide row into ten 32-trit jobs and adds up the answers.", + }, + { + kind: 'h', + text: 'What the board did', + }, + { + kind: 'table', + head: ['Run (AX7203, 1,144,744 baud)', 'Jobs', 'Rows bit-exact', 'Receipt tags compared?', 'Time'], + rows: [ + ['Random 320x320 ternary matvec', '3,200', '320 / 320', 'yes', '0.67 s'], + ['Layer 0 wq, trained weights', '12,800', '1,280 / 1,280', 'no', '2.7 s'], + ['wq, wo, gate, up in all six layers', '284,160', '28,416 / 28,416', 'no', '62.4 s'], + ], + }, + { + kind: 'p', + text: "The activations were ternary vectors, not the model's int8 activations. So this shows the trained weight matrices computed exactly. It is not a forward pass of the model.", + }, + { + kind: 'h', + text: 'The receipts in the last two rows were never compared', + }, + { + kind: 'p', + text: "The layer harness built each receipt's preimage and then dropped it, and the key it loaded was never used. It counted status byte 0x01 as authentication. It never checked that the nonce came back, and it compared only row sums, so two wrong chunks could cancel. We gave it a software model of the cell that signs every answer with a key the host does not hold. It printed 'receipts authenticated 160/160' and PASS.", + }, + { + kind: 'p', + text: "What stands: every row value in the table matched the CPU oracle, and the random-matrix run did compare its tags. What is withdrawn: '284,160 receipts authenticated under node0's key', because those receipts were never checked. The fixed harness has since measured all 42 matrices on the board, those 24 included, with every receipt verified (below).", + }, + { + kind: 'h', + text: 'The fixed harness, and seven ways to make it fail', + }, + { + kind: 'p', + text: "A response now counts only if all five of these hold. Its status is 0x01. Its nonce was issued by this run and answered once. Its node id matches the first answer. Its SipHash tag recomputes under the key. Its y equals that chunk's dot product. A row passes only if every chunk passed and their sum equals the row dot computed directly from the model's int8 weights, not from the packed wire bytes. Each check is shown able to fail:", + }, + { + kind: 'table', + head: ['What the cell does wrong', 'Harness verdict'], + rows: [ + ['signs with another key', 'rejected: tag'], + ['returns a wrong answer and signs it validly', 'rejected: lie'], + ['flips one tag bit', 'rejected: tag'], + ['answers with a nonce the run never issued', 'rejected: fabricated'], + ['answers as another node id', 'rejected: node'], + ['drops a response', 'rejected: short read, run stops'], + ['was never keyed (status 0x04)', 'no credit, though y is right'], + ], + }, + { + kind: 'p', + text: "After the board run below, an eighth control was added, for the link rather than the cell. It is a stream that loses 16 bytes in the middle, as the board's link did. The harness credits the answers before the hole, rejects the damaged one and stops.", + }, + { + kind: 'h', + text: 'The same bytes through the RTL', + }, + { + kind: 'p', + text: "The cell's own source, trinet_node_core.v and trinet_siphash24.v, unchanged, was simulated in Icarus Verilog at UART bit level. The key was installed over the wire exactly as on the board. The weights are random ternary values in tern_tc's exact shapes, because the trained model file is not in this environment.", + }, + { + kind: 'table', + head: ['RTL run', 'Jobs', 'Rows bit-exact', 'Receipts verified'], + rows: [ + ['Layer 0, all seven matrices, w_down as 27 chunks', '67,200', '5,632 / 5,632', '67,200 / 67,200'], + ['Layer 5 w_down, int8 activations (6 digit planes)', '51,840', '320 / 320', '51,840 / 51,840'], + ['Layer 0 wk, node never keyed', '640', '0 / 64', '0 / 640 (status 0x04)'], + ['Layer 0 wk, checked under a different key', '1,280', '0 / 128', '0 / 1,280'], + ], + }, + { + kind: 'h', + text: 'The fixed harness on the board', + }, + { + kind: 'p', + text: "On 2026-09-27 the fixed harness ran on the AX7203 with the trained model file. The host was an M1 Pro, talking to the board's CP2102N UART through a USB hub. The x-vectors are random test vectors from a fixed seed, not the model's real activations.", + }, + { + kind: 'table', + head: ['Board run (1,144,744 baud)', 'Jobs in flight', 'Receipts verified', 'Rows bit-exact', 'Result'], + rows: [ + ['Random 320x320 ternary matvec', '64', '3,200 / 3,200', '320 / 320', 'PASS, 0.79 s'], + ['All 42 matrices, ternary x', '64', '18,984 of 403,200, then the link lost 16 bytes', '1,898 / 33,792', 'FAIL, stopped'], + ['Layer 5 w_down, int8 x', '64', '9,886 of 51,840, then the link lost 4 bytes', '61 / 320', 'FAIL, stopped'], + ['Layer 5 w_down, int8 x', '8', '51,840 / 51,840', '320 / 320', 'PASS, 11.9 s'], + ['All 42 matrices, ternary x', '24', '403,200 / 403,200', '33,792 / 33,792', 'PASS, 85.5 s'], + ['All 42 matrices, ternary x (control)', '64', '6,679 of 403,200, then the link lost 59 bytes', '667 / 33,792', 'FAIL, stopped'], + ['All 42 matrices, ternary x (registered)', '26', '403,200 / 403,200', '33,792 / 33,792', 'PASS, 85.3 s'], + ['All 42 matrices, ternary x (registered)', '30', '149,986 of 403,200, then the link lost 7 bytes', '12,822 / 33,792', 'FAIL, stopped'], + ], + }, + { + kind: 'p', + text: "The window-24 row is the result this post was waiting for. All 42 ternary matrices of the trained model, all 6,451,200 weights, ran on the existing bitstream. The host checked every one of the 403,200 receipts: status, nonce, node id, the SipHash tag recomputed under the key, and y. All 33,792 rows equal the int8-weight oracle. The run took 85.5 s, 4,716 answers per second.", + }, + { + kind: 'p', + text: "The w_down run at 8 in flight is the first board run of trained weights with int8 activations and every receipt checked: 51,840 tags and 320 rows. It covers one matrix of 42, with one random activation vector.", + }, + { + kind: 'p', + text: "The four failures are not wrong answers. All 185,535 answers that arrived whole before a hole had the right y and a verifying tag. The one answer each hole tore through failed a check and was refused. The link is what failed. The answer stream lost 16, 4, 59 and 7 bytes, with intact bytes on both sides. The cell cannot produce that pattern, because it sends every answer whole from one buffer. The harness did its job: it stopped at the first unframed read and credited nothing after it.", + }, + { + kind: 'p', + text: "The window matters; the control shows it. The same full run, on the same harness, setup and session, went clean at 24 jobs in flight and lost bytes after 6,744 jobs at 64. So the pass comes from the window, not from the new harness's timing and hex output. At 64, three long runs slipped three times.", + }, + { + kind: 'p', + text: "The first explanation does not survive the control. The idea was that answers pile up while the harness process looks away. The fixed harness measures those pauses. In the failing control at 64 the longest was 4.2 ms, about 1,700 answers before the hole. Passing runs survived pauses of 22.4 ms at 24 in flight and 25.8 ms at 26.", + }, + { + kind: 'p', + text: "What does fit is the USB bridge. The board's UART goes to the host through a CP2102N, whose datasheet gives a 512-byte receive buffer and asks for handshaking above 1 Mbaud to avoid receiver overrun. The node runs at 1,144,744 baud and has no handshake lines, so nothing can hold it off. With W jobs in flight, at most 19 x W answer bytes are on their way. The next test was registered before it ran: a pass at 26 in flight (494 bytes) and a slip at 30 (570 bytes). Both happened. Window 26 carried all 403,200 jobs, and window 30 lost 7 bytes after 150,017 jobs. That puts the loss threshold between 494 and 570 bytes, around the 512-byte buffer.", + }, + { + kind: 'p', + text: "One side prediction missed. The slip at 30 came after 150,017 jobs, not within the first 20,000 as at 64, so the stalls that fill the buffer are rarer than a single fixed gap would explain. What stalls the bridge's USB transfers is not measured, and the hub is not excluded. The rule the harness follows now is to keep 19 x W under 512 bytes, which is why the default is 24 (456 bytes). It costs no measurable throughput: 4,716 answers/s, against about 4,660 to 4,750 at 64 before the slips.", + }, + { + kind: 'h', + text: 'Two items booked as new hardware need none', + }, + { + kind: 'p', + text: 'w_down was left out because its input is 864 wide and the plan called for a wider cell. It does not need one: 864 = 27 x 32, so it is 27 jobs per row on the same cell. wk and wv have 320-wide inputs and were simply skipped. Together that is all 42 ternary matrices and all 6,451,200 weights on the existing bitstream.', + }, + { + kind: 'p', + text: 'int8 activations were Stage B.3, which planned new RTL. Every integer from -364 to 364 is a sum of six balanced-ternary digits: q = sum of 3^k d_k with each d_k in {-1, 0, +1}. So w.q = sum of 3^k (w.d_k), and each w.d_k is an ordinary ternary job. Six jobs per chunk replace a new datapath. The cell stays ternary, and the host applies the powers of three in the open.', + }, + { + kind: 'h', + text: 'What this is not', + }, + { + kind: 'ul', + items: [ + 'Not a forward pass. Embeddings, norms, attention, the softmax and the head run nowhere on the board.', + 'Not fast. At the 4,716 jobs/s measured in the full run, one token would take about 43 s with ternary activations (201,600 jobs) and about 256 s with int8 (1,209,600 jobs). Both times are derived, not measured. The UART makes this a verification instrument, not an inference engine.', + "Not a public proof. SipHash is a shared-key MAC: it tells the key holder which node answered. A third party cannot check a receipt with it, and it does not stop an operator forging their own.", + ], + }, + { + kind: 'h', + text: 'Next', + }, + { + kind: 'ol', + items: [ + 'Lift the link limit: add RTS/CTS flow control to the node, as the CP2102N datasheet asks above 1 Mbaud. Until then, keep answers in flight under 512 bytes, as the harness now does.', + 'Real activations: dump the int8 inputs tc_infer computes for a real prompt, and run a whole layer with them.', + 'Leave the UART: move to Ethernet or a USB FIFO, and measure it on the board.', + 'Make receipts checkable by anyone: publish a Merkle root of each run, and add random re-execution or Freivalds checks. Both are cheap for ternary matvecs.', + 'Measure power on a bench supply. Without it there is no energy comparison to publish.', + ], + }, +] + +export const ruBody: Block[] = [ + { + kind: 'p', + text: 'Модель IGLA размером с плату, tern_tc, — это 9,08M параметров. Из них 6 451 200 — тернарные веса блоков: 42 матрицы в шести слоях. Модель обучена на 2,0 млрд токенов кода до 0,7613 бит на байт на валидации. Плата — ALINX AX7203 (XC7A200T) с ячейкой узла TRI-NET, собранной открытым тулчейном openXC7. Ячейка считает тернарное скалярное произведение на 32 трита без умножителя и без DSP48. На каждую задачу она отвечает результатом, nonce задачи, id узла и тегом SipHash-2-4 по всему этому. Хост режет каждую строку шириной 320 на десять задач по 32 трита и складывает ответы.', + }, + { + kind: 'h', + text: 'Что сделала плата', + }, + { + kind: 'table', + head: ['Прогон (AX7203, 1 144 744 бод)', 'Задач', 'Строк бит-точно', 'Теги квитанций сверялись?', 'Время'], + rows: [ + ['Случайный тернарный matvec 320x320', '3 200', '320 / 320', 'да', '0,67 с'], + ['Слой 0, wq, обученные веса', '12 800', '1 280 / 1 280', 'нет', '2,7 с'], + ['wq, wo, gate, up во всех шести слоях', '284 160', '28 416 / 28 416', 'нет', '62,4 с'], + ], + }, + { + kind: 'p', + text: 'Активации были тернарными векторами, а не int8-активациями модели. Это показывает, что обученные матрицы весов считаются точно. Прямым проходом модели это не является.', + }, + { + kind: 'h', + text: 'Квитанции в двух последних строках никто не сверял', + }, + { + kind: 'p', + text: 'Харнесс слоя собирал прообраз каждой квитанции и выбрасывал его, а загруженный ключ не использовался вовсе. Аутентификацией он считал байт статуса 0x01. Возврат nonce он не проверял и сравнивал только суммы строк, так что две ошибки в кусках могли взаимно погаситься. Мы подключили к нему программную модель ячейки, которая подписывает каждый ответ ключом, которого у хоста нет. Он напечатал «receipts authenticated 160/160» и PASS.', + }, + { + kind: 'p', + text: 'Что остаётся в силе: каждое значение строк в таблице совпало с CPU-оракулом, а прогон случайной матрицы свои теги сверял. Что отозвано: «284 160 квитанций аутентифицировано под ключом node0», потому что эти квитанции никто не проверял. С тех пор исправленный харнесс измерил на плате все 42 матрицы, включая эти 24, с проверкой каждой квитанции (ниже).', + }, + { + kind: 'h', + text: 'Исправленный харнесс и семь способов его завалить', + }, + { + kind: 'p', + text: 'Теперь ответ засчитывается, только если выполнены все пять условий. Статус равен 0x01. Nonce выдан этим прогоном, и ответ на него пришёл один раз. Id узла совпадает с первым ответом. Тег SipHash пересчитывается под ключом. y равен скалярному произведению этого куска. Строка проходит, только если прошли все её куски и их сумма равна скалярному произведению строки, посчитанному прямо по int8-весам модели, а не по упакованным байтам провода. Для каждой проверки показано, что она умеет падать:', + }, + { + kind: 'table', + head: ['Что ячейка делает не так', 'Вердикт харнесса'], + rows: [ + ['подписывает чужим ключом', 'отвергнуто: tag'], + ['отдаёт неверный ответ с корректной подписью', 'отвергнуто: lie'], + ['портит один бит тега', 'отвергнуто: tag'], + ['отвечает nonce, которого прогон не выдавал', 'отвергнуто: fabricated'], + ['отвечает от имени другого узла', 'отвергнуто: node'], + ['теряет ответ', 'отвергнуто: short read, прогон останавливается'], + ['ключ так и не установлен (статус 0x04)', 'не засчитано, хотя y верный'], + ], + }, + { + kind: 'p', + text: 'После прогона на плате, описанного ниже, добавлена восьмая проверка, уже для канала, а не для ячейки. Это поток, из середины которого пропадают 16 байт, как было в канале платы. Харнесс засчитывает ответы до дыры, отвергает повреждённый ответ и останавливается.', + }, + { + kind: 'h', + text: 'Те же байты через RTL', + }, + { + kind: 'p', + text: 'Собственные исходники ячейки, trinet_node_core.v и trinet_siphash24.v, без изменений, симулированы в Icarus Verilog на уровне битов UART. Ключ ставился по проводу ровно так же, как на плате. Веса — случайные тернарные значения точно в формах tern_tc, потому что файла обученной модели в этом окружении нет.', + }, + { + kind: 'table', + head: ['Прогон RTL', 'Задач', 'Строк бит-точно', 'Квитанций проверено'], + rows: [ + ['Слой 0, все семь матриц, w_down по 27 кускам', '67 200', '5 632 / 5 632', '67 200 / 67 200'], + ['Слой 5, w_down, int8-активации (6 разрядов)', '51 840', '320 / 320', '51 840 / 51 840'], + ['Слой 0, wk, ключ не установлен', '640', '0 / 64', '0 / 640 (статус 0x04)'], + ['Слой 0, wk, проверка под другим ключом', '1 280', '0 / 128', '0 / 1 280'], + ], + }, + { + kind: 'h', + text: 'Исправленный харнесс на плате', + }, + { + kind: 'p', + text: '27 сентября 2026 года исправленный харнесс прогнали на AX7203 с файлом обученной модели. Хостом был M1 Pro, связанный с UART платы (CP2102N) через USB-хаб. Векторы x — случайные тестовые векторы из фиксированного seed, а не настоящие активации модели.', + }, + { + kind: 'table', + head: ['Прогон на плате (1 144 744 бод)', 'Задач в полёте', 'Квитанций проверено', 'Строк бит-точно', 'Итог'], + rows: [ + ['Случайный тернарный matvec 320x320', '64', '3 200 / 3 200', '320 / 320', 'PASS, 0,79 с'], + ['Все 42 матрицы, тернарный x', '64', '18 984 из 403 200, потом канал потерял 16 байт', '1 898 / 33 792', 'FAIL, остановлен'], + ['Слой 5, w_down, int8 x', '64', '9 886 из 51 840, потом канал потерял 4 байта', '61 / 320', 'FAIL, остановлен'], + ['Слой 5, w_down, int8 x', '8', '51 840 / 51 840', '320 / 320', 'PASS, 11,9 с'], + ['Все 42 матрицы, тернарный x', '24', '403 200 / 403 200', '33 792 / 33 792', 'PASS, 85,5 с'], + ['Все 42 матрицы, тернарный x (контроль)', '64', '6 679 из 403 200, потом канал потерял 59 байт', '667 / 33 792', 'FAIL, остановлен'], + ['Все 42 матрицы, тернарный x (зарегистрирован)', '26', '403 200 / 403 200', '33 792 / 33 792', 'PASS, 85,3 с'], + ['Все 42 матрицы, тернарный x (зарегистрирован)', '30', '149 986 из 403 200, потом канал потерял 7 байт', '12 822 / 33 792', 'FAIL, остановлен'], + ], + }, + { + kind: 'p', + text: 'Строка с окном 24 — результат, которого ждал этот пост. Все 42 тернарные матрицы обученной модели, все 6 451 200 весов, отработали на текущем битстриме. Хост проверил каждую из 403 200 квитанций: статус, nonce, id узла, тег SipHash, пересчитанный под ключом, и y. Все 33 792 строки совпали с оракулом по int8-весам. Прогон занял 85,5 с, 4 716 ответов в секунду.', + }, + { + kind: 'p', + text: 'Прогон w_down при 8 задачах в полёте — первый прогон обученных весов с int8-активациями на плате, в котором проверена каждая квитанция: 51 840 тегов и 320 строк. Это одна матрица из 42 и один случайный вектор активаций.', + }, + { + kind: 'p', + text: 'Все четыре провала — не неверные ответы. У всех 185 535 ответов, пришедших целыми до дыры, был верный y и сходящийся тег. Ответ, который рвала каждая дыра, не прошёл проверку и был отвергнут. Сломался канал. Из потока ответов пропадало 16, 4, 59 и 7 байт, и по обе стороны дыры байты целые. Ячейка такую картину дать не может: каждый ответ она отправляет целиком из одного буфера. Харнесс сделал то, что должен: остановился на первом нераспознанном кадре и после него ничего не засчитал.', + }, + { + kind: 'p', + text: 'Окно имеет значение, и это показал контроль. Тот же полный прогон на том же харнессе, установке и сессии прошёл чисто при 24 задачах в полёте и потерял байты после 6 744 задач при 64. Значит, успех дало окно, а не замер времени и вывод hex, добавленные в новый харнесс. При 64 три длинных прогона сорвались три раза.', + }, + { + kind: 'p', + text: 'Первое объяснение контроль не выдержало. Предполагалось, что ответы копятся, пока процесс харнесса отвлёкся. Исправленный харнесс эти паузы измеряет. В провальном контроле при 64 самая долгая была 4,2 мс, примерно за 1 700 ответов до дыры. Успешные прогоны пережили паузы 22,4 мс при 24 задачах в полёте и 25,8 мс при 26.', + }, + { + kind: 'p', + text: 'Зато подходит USB-мост. UART платы идёт к хосту через CP2102N; по его даташиту у него приёмный буфер на 512 байт, а выше 1 Мбод нужно аппаратное управление потоком, иначе приёмник переполняется. Узел работает на 1 144 744 бод, и линий управления потоком у него нет, так что придержать его нечем. При W задачах в полёте к хосту идёт не больше 19 x W байт ответов. Следующую проверку зарегистрировали до запуска: при 26 задачах в полёте (494 байта) — успех, при 30 (570 байт) — срыв. Так и вышло. Окно 26 провело все 403 200 задач, окно 30 потеряло 7 байт после 150 017 задач. Порог потерь лежит между 494 и 570 байтами, около буфера на 512.', + }, + { + kind: 'p', + text: 'Одно побочное предсказание не сбылось. Срыв при 30 пришёл после 150 017 задач, а не в первые 20 000, как при 64, так что задержки, заполняющие буфер, реже, чем объяснил бы один фиксированный простой. Что задерживает USB-передачи моста, не измерено, и хаб не исключён. Правило харнесса теперь — держать 19 x W меньше 512 байт, поэтому по умолчанию окно 24 (456 байт). Скорости это заметно не стоит: 4 716 ответов в секунду против примерно 4 660–4 750 при 64 до сбоев.', + }, + { + kind: 'h', + text: 'Две задачи, записанные в новое железо, его не требуют', + }, + { + kind: 'p', + text: 'w_down пропустили, потому что у неё вход шириной 864, и по плану для неё нужна была ячейка шире. Не нужна: 864 = 27 x 32, то есть это 27 задач на строку на той же ячейке. У wk и wv вход шириной 320, их просто пропустили. Вместе это все 42 тернарные матрицы и все 6 451 200 весов на текущем битстриме.', + }, + { + kind: 'p', + text: 'int8-активации числились за Stage B.3, где планировался новый RTL. Любое целое от -364 до 364 — это сумма шести сбалансированно-троичных разрядов: q = сумма 3^k d_k, где каждый d_k из {-1, 0, +1}. Значит, w.q = сумма 3^k (w.d_k), и каждое w.d_k — обычная тернарная задача. Шесть задач на кусок заменяют новый тракт данных. Ячейка остаётся тернарной, а степени тройки хост применяет открыто.', + }, + { + kind: 'h', + text: 'Чем это не является', + }, + { + kind: 'ul', + items: [ + 'Это не прямой проход. Эмбеддинги, нормы, внимание, softmax и голова на плате не выполняются.', + 'Это не быстро. При 4 716 задачах/с, измеренных в полном прогоне, один токен занял бы около 43 с с тернарными активациями (201 600 задач) и около 256 с с int8 (1 209 600 задач). Оба времени выведены, а не измерены. Из-за UART это инструмент проверки, а не движок вывода.', + 'Это не публичное доказательство. SipHash — MAC с общим ключом: он сообщает владельцу ключа, какой узел ответил. Третья сторона не может им проверить квитанцию, и он не мешает оператору подделать свои собственные.', + ], + }, + { + kind: 'h', + text: 'Дальше', + }, + { + kind: 'ol', + items: [ + 'Снять ограничение канала: добавить узлу управление потоком RTS/CTS, как требует даташит CP2102N выше 1 Мбод. До тех пор держать ответы в полёте меньше 512 байт, как теперь делает харнесс.', + 'Реальные активации: выгрузить int8-входы, которые tc_infer считает для настоящего промпта, и прогнать с ними целый слой.', + 'Уйти с UART: перейти на Ethernet или USB FIFO и измерить это на плате.', + 'Сделать квитанции проверяемыми для всех: публиковать корень Меркла каждого прогона и добавить случайное перевычисление или проверки Фрейвалдса. Для тернарных matvec и то и другое стоит дёшево.', + 'Измерить мощность от лабораторного блока питания. Без этого публиковать сравнение по энергии нечем.', + ], + }, +] diff --git a/apps/website/src/data/blog/index.ts b/apps/website/src/data/blog/index.ts index 7100ea79e3..9cdb8825ef 100644 --- a/apps/website/src/data/blog/index.ts +++ b/apps/website/src/data/blog/index.ts @@ -2,10 +2,84 @@ import type { PostMeta } from './types' /** Индекс блога: список и метаданные без тяжёлых тел публикаций. */ export const postsIndex: PostMeta[] = [ + { + slug: "golden-ratio-weights-ran-on-the-board", + title: "Golden-ratio weights ran on the board. The phi cost one add per output.", + summary: "[measured on FPGA: layer 0, 403,200 of 403,200 receipts verified, registered before the run; the phi step and digit recombination are host arithmetic; one synthetic activation vector] The TNF paper's GFTernary weights, t*phi, applied to activations in Z[phi], ran on the unchanged AX7203 node with no multiplier. Each output is W.b + (W.a + W.b)*phi, so the node computes only ternary dots and the host adds once per output. All 5,632 Z[phi] rows of layer 0 matched a plain Z[phi] multiplication oracle bit for bit.", + date: "2026-09-27", + readingMinutes: 6, + tags: ["FPGA", "Ternary", "Number formats", "Verification", "TRI-NET"], + receipts: [ + { label: "The pre-registration and the result, side by side", href: "https://github.com/gHashTag/trinity-fpga/blob/8a203217084eb3499d6dee5815be6a462575878f/conformance/GFT_ZPHI_ON_THE_NODE.md" }, + { label: "The board log: 403,200 / 403,200 receipts, 5,632 / 5,632 Z[phi] rows", href: "https://github.com/gHashTag/trinity-fpga/blob/8a203217084eb3499d6dee5815be6a462575878f/conformance/board_runs/gft_zphi_l0.log" }, + { label: "The harness, with its Z[phi] oracle and negative controls", href: "https://github.com/gHashTag/trinity-fpga/blob/8a203217084eb3499d6dee5815be6a462575878f/conformance/gft_zphi_ax7203.py" }, + { label: "Ten weak points of the tern_tc record, ranked", href: "https://github.com/gHashTag/trinity-fpga/blob/8a203217084eb3499d6dee5815be6a462575878f/conformance/TERN_TC_WEAK_POINTS.md" }, + { label: "The pull request that carries it: gHashTag/trinity-fpga 800", href: "https://github.com/gHashTag/trinity-fpga/pull/800" }, + { label: "Earlier: the golden ratio in this format is a scale factor, not information", href: "https://t27.ai/blog/phi-is-a-scale-not-information/" }, + { label: "Earlier: the board's receipts were never checked; now all 403,200 are", href: "https://t27.ai/blog/trained-weights-ran-receipts-were-not-checked/" }, + ], + openQuestions: [ + "Only layer 0 ran, with one synthetic Z[phi] activation vector, not the activations of a real forward pass.", + "The TNF accumulator and its rounding have no RTL, so the part of the claim beyond the dot products is untested on hardware.", + "The node computes ternary dots; the digit recombination and the phi step run on the host. The result says the linear path needs no multiplier, not that the FPGA did all of it.", + "Speed is not claimed: about 151,000 ternary multiply-accumulates per second over UART (derived).", + "A SipHash receipt is not publicly verifiable, and it does not stop an operator forging their own.", + ], + published: true, + ru: { + title: "Веса с золотым сечением отработали на плате. phi стоил одно сложение на выход.", + summary: "[измерено на FPGA: слой 0, проверено 403 200 из 403 200 квитанций, зарегистрировано до запуска; шаг phi и сборка разрядов — арифметика хоста; один синтетический вектор активаций] Веса GFTernary из статьи о TNF, t*phi, применённые к активациям из Z[phi], отработали на неизменённом узле AX7203 без умножителя. Каждый выход равен W.b + (W.a + W.b)*phi, так что узел считает только тернарные скалярные произведения, а хост делает одно сложение на выход. Все 5 632 строки Z[phi] слоя 0 совпали бит в бит с оракулом обычного умножения в Z[phi].", + openQuestions: [ + "Прогнан только слой 0 с одним синтетическим вектором активаций из Z[phi], а не с активациями настоящего прямого прохода.", + "У аккумулятора TNF и его округления нет RTL, так что часть утверждения сверх скалярных произведений на железе не проверена.", + "Узел считает тернарные скалярные произведения; сборка разрядов и шаг phi выполняются на хосте. Результат говорит, что линейному пути не нужен умножитель, а не что всё сделала FPGA.", + "Скорость не заявляется: около 151 000 тернарных умножений с накоплением в секунду через UART (выведено).", + "Квитанция SipHash не проверяется публично и не мешает оператору подделать свои собственные.", + ], + }, + }, + { + slug: "trained-weights-ran-receipts-were-not-checked", + title: "The board's receipts were never checked. Now all 403,200 are.", + summary: "[measured on FPGA: all 42 trained matrices, every receipt verified; the activations were test vectors, not a forward pass] An AX7203 computed 28,416 of 28,416 rows of the trained tern_tc model's weight matrices bit-exact, while its harness reported 284,160 authenticated receipts without comparing a single tag. The fixed harness fails on seven kinds of misbehaving cell. On the board it then verified all 403,200 receipts across all 42 ternary matrices, with 33,792 of 33,792 rows bit-exact, plus 51,840 for w_down with int8 activations. Its first long runs stopped when a UART link dropped bytes with 64 jobs in flight. A test registered before it ran put the loss threshold at the USB bridge's 512-byte receive buffer, and at 24 in flight none were lost.", + date: "2026-09-27", + readingMinutes: 11, + tags: ["IGLA", "FPGA", "Ternary", "Verification", "TRI-NET", "Correction"], + receipts: [ + { label: "The fix: gHashTag/trinity-fpga pull request 800", href: "https://github.com/gHashTag/trinity-fpga/pull/800" }, + { label: "The commit whose receipt count is withdrawn (1f131fd)", href: "https://github.com/gHashTag/trinity-fpga/commit/1f131fdd3fdf1c78317a923a7cd070e11f101a8a" }, + { label: "What is verified, by what, and how to reproduce it; every board run, the controls, the registered buffer test and the byte analysis", href: "https://github.com/gHashTag/trinity-fpga/blob/09b3d6f733e4fe2df8aa09791db1ecde8b2f54c6/conformance/TERN_TC_LAYER_RECEIPTS.md" }, + { label: "The full 42-matrix board log: 403,200 / 403,200 receipts verified", href: "https://github.com/gHashTag/trinity-fpga/blob/09b3d6f733e4fe2df8aa09791db1ecde8b2f54c6/conformance/board_runs/tern_tc_all_w24.log" }, + { label: "The window-64 control log: the link lost 59 bytes after 6,744 jobs", href: "https://github.com/gHashTag/trinity-fpga/blob/09b3d6f733e4fe2df8aa09791db1ecde8b2f54c6/conformance/board_runs/tern_tc_all_w64.log" }, + { label: "The registered buffer test: window 26 (494 B) clean, window 30 (570 B) slipped", href: "https://github.com/gHashTag/trinity-fpga/blob/09b3d6f733e4fe2df8aa09791db1ecde8b2f54c6/conformance/board_runs/tern_tc_all_w30.log" }, + { label: "All raw board logs from 2026-09-27", href: "https://github.com/gHashTag/trinity-fpga/tree/09b3d6f733e4fe2df8aa09791db1ecde8b2f54c6/conformance/board_runs" }, + { label: "Prior art for tern_tc, with the CP2102N figures checked against its datasheet", href: "https://github.com/gHashTag/trinity-fpga/blob/09b3d6f733e4fe2df8aa09791db1ecde8b2f54c6/conformance/TERN_TC_PRIOR_ART.md" }, + { label: "The fixed harness that ran them, with its negative controls", href: "https://github.com/gHashTag/trinity-fpga/blob/b1e95f6fd5ec43b5be709af86b179750ae8b27a5/conformance/tern_tc_layer_ax7203.py" }, + { label: "The RTL testbench that replays the harness over UART", href: "https://github.com/gHashTag/trinity-fpga/blob/2b9830c5837aa47ad142a90a7279c2d8a8d401fe/formal/tern_tc_layer_rtl_tb.v" }, + { label: "The node cell under test: trinet_node_core.v", href: "https://github.com/gHashTag/trinity-fpga/blob/2b9830c5837aa47ad142a90a7279c2d8a8d401fe/fpga/portable/trinet_node_core.v" }, + ], + openQuestions: [ + "The loss threshold sits between 494 and 570 bytes in flight, around the CP2102N's 512-byte receive buffer, and the node has no flow control. What stalls the bridge's USB transfers long enough to fill it is not measured, and the hub is not excluded.", + "The activation vectors were random test vectors, not the model's real activations. int8 activations have run on the board for one matrix.", + "No forward pass, token rate or power figure has been measured on the board.", + "A SipHash receipt is not publicly verifiable, and it does not stop an operator forging their own.", + ], + published: true, + ru: { + title: "Квитанции с платы никто не проверял. Теперь проверены все 403 200.", + summary: "[измерено на FPGA: все 42 обученные матрицы, проверена каждая квитанция; активации — тестовые векторы, а не прямой проход] AX7203 посчитала 28 416 из 28 416 строк матриц весов обученной модели tern_tc бит-точно, а её харнесс сообщил о 284 160 аутентифицированных квитанциях, не сравнив ни одного тега. Исправленный харнесс падает на семи видах неисправной ячейки. На плате он затем проверил все 403 200 квитанций по всем 42 тернарным матрицам, 33 792 из 33 792 строк бит-точно, и ещё 51 840 для w_down с int8-активациями. Первые длинные прогоны остановились, когда канал UART терял байты при 64 задачах в полёте. Проверка, зарегистрированная до запуска, поместила порог потерь на приёмный буфер USB-моста в 512 байт, а при 24 задачах в полёте не потерялось ни одного байта.", + openQuestions: [ + "Порог потерь лежит между 494 и 570 байтами в полёте, около приёмного буфера CP2102N в 512 байт, а у узла нет управления потоком. Что задерживает USB-передачи моста настолько, чтобы буфер заполнился, не измерено, и хаб не исключён.", + "Векторы активаций были случайными тестовыми, а не настоящими активациями модели. int8-активации на плате прогнаны для одной матрицы.", + "Ни прямой проход, ни скорость в токенах, ни мощность на плате не измерены.", + "Квитанция SipHash не проверяется публично и не мешает оператору подделать свои собственные.", + ], + }, + }, { slug: "a-small-agent-needs-an-exact-judge", title: "A small agent is only useful next to an exact judge", - summary: "[model results measured on GPUs; board throughput derived, not measured] One Artix-7 200T holds the layers of a 13M-parameter ternary model in its own block memory. In our own twin experiment the 100M ternary model passes 0.97% vs 1.68% for full precision across eight languages, and compile rate, not correctness, is the gap. A model that is wrong most of the time becomes useful only where a compiler checks every answer.", + summary: "[model results measured on GPUs; board throughput derived, not measured] One Artix-7 200T holds the layers of a 13M-parameter ternary model in its own block memory. In our own twin experiment the 100M ternary model passes 1.01% vs 1.68% for full precision across eight languages, and compile rate, not correctness, is the gap. A model that is wrong most of the time becomes useful only where a compiler checks every answer.", date: "2026-09-26", readingMinutes: 9, tags: ["IGLA", "Agents", "FPGA", "Ternary", "Verification", "Plan"], @@ -21,24 +95,24 @@ export const postsIndex: PostMeta[] = [ { label: "MultiPL-E: MultiPL-E benchmark, HumanEval-164 in 18+ languages (arXiv:2208.08227)", href: "https://arxiv.org/abs/2208.08227" }, ], openQuestions: [ - "No IGLA model has run on a board. Every throughput figure here is a ceiling derived from memory bandwidth and LUT counts.", + "Update 2026-09-27: the trained tern_tc model's 320-input weight matrices ran bit-exact on an AX7203 (see the follow-up post). No forward pass has run on a board, and every throughput figure here is still a ceiling derived from memory bandwidth and LUT counts.", "The code ability of a 13M model on .t27 is unknown. The HumanEval figures above are for Python and for another model family.", "The open DDR3 path on Artix-7 has passed memtest on hardware only since nextpnr-xilinx 0.9.5 (13 September 2026). The independent UberDDR3 test reached a 333 MHz DDR clock, not the 400 MHz the AX7203 is rated for.", "Integer inference must be shown to cost little quality against the float model before receipts can rest on it.", "The c-BTM experts had 1.3B parameters or more, a hundred times the size of a board-sized one. Whether the result holds at 13M is a measurement, not an inference.", - "The twin numbers are for 100M-parameter models trained on GPUs; the board-sized 13M ternary model has been neither trained nor run on a board.", + "The twin numbers are for 100M-parameter models trained on GPUs. A board-sized ternary model (tern_tc, 9.08M parameters) has since been trained; its code ability has not been measured.", ], published: true, ru: { title: "Маленький агент полезен только рядом с точным судьёй", - summary: "[результаты моделей измерены на GPU; пропускная способность платы выведена, не измерена] Одна Artix-7 200T держит слои тернарной модели на 13M параметров в собственной блочной памяти. В нашем парном эксперименте тернарная модель на 100M проходит 0,97% против 1,68% у полной точности на восьми языках, и разрыв — это компиляция, а не корректность. Модель, чаще ошибающаяся, полезна только там, где каждый ответ проверяет компилятор.", + summary: "[результаты моделей измерены на GPU; пропускная способность платы выведена, не измерена] Одна Artix-7 200T держит слои тернарной модели на 13M параметров в собственной блочной памяти. В нашем парном эксперименте тернарная модель на 100M проходит 1,01% против 1,68% у полной точности на восьми языках, и разрыв — это компиляция, а не корректность. Модель, чаще ошибающаяся, полезна только там, где каждый ответ проверяет компилятор.", openQuestions: [ - "Ни одна модель IGLA ещё не запускалась на плате. Каждая цифра скорости здесь — потолок, выведенный из пропускной способности памяти и числа LUT.", + "Обновление 2026-09-27: матрицы весов с входом 320 обученной модели tern_tc посчитались на AX7203 бит-точно (см. следующий пост). Прямой проход на плате не запускался, и каждая цифра скорости здесь по-прежнему потолок, выведенный из пропускной способности памяти и числа LUT.", "Способность модели на 13M писать .t27 неизвестна. Цифры HumanEval выше относятся к Python и к другому семейству моделей.", "Открытый путь к DDR3 на Artix-7 проходит memtest на железе только с nextpnr-xilinx 0.9.5 (13 сентября 2026). Независимый тест UberDDR3 достиг частоты DDR 333 МГц, а не 400 МГц, на которые рассчитана AX7203.", "Нужно показать, что целочисленный вывод почти не теряет в качестве по сравнению с плавающей точкой. Только после этого на него могут опираться квитанции.", "Эксперты в c-BTM имели 1,3B параметров и больше — в сто раз больше, чем модель размером с плату. Держится ли результат при 13M, решит замер, а не рассуждение.", - "Числа двойняшек относятся к моделям на 100M параметров, обученным на GPU; модель на 13M, помещающаяся на плату, не обучена и не запускалась на плате.", + "Числа двойняшек относятся к моделям на 100M параметров, обученным на GPU. С тех пор обучена тернарная модель размером с плату (tern_tc, 9,08M параметров); её способность писать код не измерена.", ], }, }, diff --git a/apps/website/src/data/blog/posts.ts b/apps/website/src/data/blog/posts.ts index c2fd894780..26789ca811 100644 --- a/apps/website/src/data/blog/posts.ts +++ b/apps/website/src/data/blog/posts.ts @@ -17,6 +17,8 @@ import { body as body_the_fpga_row_was_corrected, ruBody as ruBody_the_fpga_row_ import { body as body_a_small_agent_needs_an_exact_judge, ruBody as ruBody_a_small_agent_needs_an_exact_judge } from './bodies/a-small-agent-needs-an-exact-judge' import type { Post, PostBody } from './types' import { body as body_tri_mined_not_sold, ruBody as ruBody_tri_mined_not_sold } from './bodies/tri-mined-not-sold' +import { body as body_trained_weights_ran_receipts_were_not_checked, ruBody as ruBody_trained_weights_ran_receipts_were_not_checked } from './bodies/trained-weights-ran-receipts-were-not-checked' +import { body as body_golden_ratio_weights_ran_on_the_board, ruBody as ruBody_golden_ratio_weights_ran_on_the_board } from './bodies/golden-ratio-weights-ran-on-the-board' import { body as body_the_only_stable_speed_belonged_to_the_tool, ruBody as ruBody_the_only_stable_speed_belonged_to_the_tool } from './bodies/the-only-stable-speed-belonged-to-the-tool' import { body as body_queen_review_lifecycle_queues, ruBody as ruBody_queen_review_lifecycle_queues } from './bodies/queen-review-lifecycle-queues' import { body as body_physical_width_changed_the_question, ruBody as ruBody_physical_width_changed_the_question } from './bodies/physical-width-changed-the-question' @@ -148,6 +150,8 @@ const bodies: Record = { 'a-multiplicity-correction-changed-the-deployment-reading': { body: body_a_multiplicity_correction_changed_the_deployment_reading, ruBody: ruBody_a_multiplicity_correction_changed_the_deployment_reading }, 'the-only-stable-speed-belonged-to-the-tool': { body: body_the_only_stable_speed_belonged_to_the_tool, ruBody: ruBody_the_only_stable_speed_belonged_to_the_tool }, 'a-small-agent-needs-an-exact-judge': { body: body_a_small_agent_needs_an_exact_judge, ruBody: ruBody_a_small_agent_needs_an_exact_judge }, + 'trained-weights-ran-receipts-were-not-checked': { body: body_trained_weights_ran_receipts_were_not_checked, ruBody: ruBody_trained_weights_ran_receipts_were_not_checked }, + 'golden-ratio-weights-ran-on-the-board': { body: body_golden_ratio_weights_ran_on_the_board, ruBody: ruBody_golden_ratio_weights_ran_on_the_board }, } export type { Block, Post } from './types' diff --git a/apps/website/src/lib/queenWars.generated.ts b/apps/website/src/lib/queenWars.generated.ts index e15d9bff1e..8e3366ceb1 100644 --- a/apps/website/src/lib/queenWars.generated.ts +++ b/apps/website/src/lib/queenWars.generated.ts @@ -1,12 +1,12 @@ // GENERATED by scripts/queen-wars-from-spec.mjs from specs/queen/wars.t27 -// spec sha256 fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8 +// spec sha256 43e610413821968b743749cb01bed33a54eb9c3b05203b16035c2be2128da5d6 // Do not edit: change the .t27 source and regenerate. export const QUEEN_WARS = { "source": { "spec": "specs/queen/wars.t27", "publicSpec": "public/queen/wars.t27", - "sha256": "fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8", + "sha256": "43e610413821968b743749cb01bed33a54eb9c3b05203b16035c2be2128da5d6", "schemaVersion": 2 }, "name": "Queen WARS real-task agent arena", @@ -28,6 +28,7 @@ export const QUEEN_WARS = { "planned", "running", "credential-blocked", + "judged", "complete", "invalid" ], @@ -111,6 +112,18 @@ export const QUEEN_WARS = { { "key": "jev-confidence", "unit": "probability" + }, + { + "key": "total-tokens", + "unit": "tokens" + }, + { + "key": "judge-calls", + "unit": "count" + }, + { + "key": "mutants-killed", + "unit": "count" } ], "configurations": [ @@ -120,12 +133,24 @@ export const QUEEN_WARS = { "kind": "coding-agent", "state": "ready", "evidence": "OBSERVED", - "source": "current Codex task runtime", - "note": "Control arm: the coding Bee chooses and implements without an external decision model.", + "source": "coding agent runtime: Codex task runtime (2026-09-23 run); Claude Code subagents (2026-09-26 campaign)", + "note": "Control arm: the coding Bee chooses and implements without an external decision layer. In the 2026-09-26 campaign it had no t27c while it worked; the gates judged it afterwards.", "stateEvidence": "OBSERVED", - "stateSource": "current Codex task runtime on 2026-09-23", + "stateSource": "current Codex task runtime on 2026-09-23; Claude Code subagents on 2026-09-26", "stateNote": "The baseline Bee can run locally." }, + { + "id": "bee-tri", + "name": "Bee + TRI", + "kind": "coding-agent-plus-compiler-judge", + "state": "ready", + "evidence": "OBSERVED", + "source": "t27c 0.4.0 built from the gHashTag/t27 bootstrap at afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "note": "Variable arm: the same Bee, prompt, tools and budget, plus the t27c compiler as its decision layer. It may run read-only t27c subcommands to choose between candidate edits; t27c never writes the patch.", + "stateEvidence": "OBSERVED", + "stateSource": "Queen WARS campaign 2026-09-26: six paired arms on gHashTag/t27 issues 4614, 4695 and 4613", + "stateNote": "Runs wherever t27c and zig 0.16.0 build; no credential is needed." + }, { "id": "bee-jev", "name": "Bee + JEV", @@ -133,7 +158,7 @@ export const QUEEN_WARS = { "state": "credential-blocked", "evidence": "SOURCE-CLAIM", "source": "https://docs.typesafe.ai/introduction/coding-agents", - "note": "Same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", + "note": "Comparison arm only: the same coding Bee with JEV restricted to ranking typed choices; JEV does not generate the patch.", "stateEvidence": "OBSERVED", "stateSource": "credential-name audit: current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth on 2026-09-23; production Railway variables were not inspected", "stateNote": "No usable TypeSafe or Cloudflare credential was found in the audited locations. Production Railway variables were not inspected, so credential absence there is not claimed." @@ -146,9 +171,9 @@ export const QUEEN_WARS = { "evidence": "OBSERVED", "source": "https://t27.ai/t27/files/specs/igla/coder/pipeline.t27", "note": "The .t27 corpus describes the coder pipeline; this lane becomes measurable only with an executable checkpoint.", - "stateEvidence": "OBSERVED", - "stateSource": "checkpoint discovery audit of the linked IGLA .t27 pipeline on 2026-09-23", - "stateNote": "No executable checkpoint was verified for this arena." + "stateEvidence": "SOURCE-CLAIM", + "stateSource": "gHashTag/igla-coder-gpu LEDGER.md and c_infer/model.bin.json, read 2026-09-26; the repository is private", + "stateNote": "The IGLA pilot reports trained checkpoints (tern_tc: 9.08M parameters, 0.7613 bits per byte after 2.0B tokens; a 100M full-precision and ternary pair) and 2.0% and 1.3% pass@10 on a t27 fill-in-the-body bench. The checkpoints are not public, and none has run an arena issue." }, { "id": "igla-race", @@ -187,6 +212,78 @@ export const QUEEN_WARS = { "state": "credential-blocked", "evidence": "OBSERVED", "note": "This baseline is a preflight result, not one side of a future A/B comparison: model identity is unknown. When JEV authentication is connected, rerun both arms together with the same explicit model and reasoning configuration. The JEV arm is blocked in this runtime because no usable authentication was found in the audited locations; unaudited production secret state is not claimed." + }, + { + "id": "t27-4614-count-buttons", + "name": "count_buttons missing test", + "issue": { + "repo": "gHashTag/t27", + "number": 4614, + "url": "https://github.com/gHashTag/t27/issues/4614", + "updatedAt": "2026-09-23T09:31:53Z" + }, + "baseSha": "afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "executorModel": "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", + "prompt": "Resolve gHashTag/t27 issue 4614 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4614.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", + "promptSha256": "76e1a46c458f60b3ff900d49b4ac8d22a209c6aa19948573951c4b5fcf75e953", + "toolPolicy": "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", + "toolPolicySha256": "803db5972a1ba27b9139c51f5790e6f477ff1065b36fbc06331a2b166d6ea41a", + "budget": "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", + "acceptance": "t27c coverage specs/boards/arty_a7.t27 reports Untested: 0; function count remains 5; test count is at least 18; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", + "acceptanceSha256": "b677c416dec6736730e8b730528ca8a98649780f808f8c6e64eb548133202d59", + "modelEvidence": "SESSION-OBSERVED", + "modelSource": "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", + "state": "judged", + "evidence": "OBSERVED", + "note": "Both arms wrote the same 4-line test, byte for byte, and both kill 3 of 3 mutants of count_buttons. There is no difference to rank." + }, + { + "id": "t27-4695-tick", + "name": "tick missing test", + "issue": { + "repo": "gHashTag/t27", + "number": 4695, + "url": "https://github.com/gHashTag/t27/issues/4695", + "updatedAt": "2026-09-24T02:32:52Z" + }, + "baseSha": "afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "executorModel": "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", + "prompt": "Resolve gHashTag/t27 issue 4695 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4695.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", + "promptSha256": "720c9cfcc51101e421d3941e0840f30484d2700aa71aa405a7b7707e7a89c2e5", + "toolPolicy": "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", + "toolPolicySha256": "803db5972a1ba27b9139c51f5790e6f477ff1065b36fbc06331a2b166d6ea41a", + "budget": "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", + "acceptance": "t27c coverage specs/fpga/testbench/simulator_tb.t27 reports Untested: 0; function count remains 4; test count is at least 7; t27c spec-status reports IMPLEMENTED; t27c test-report reports 0 BLOCKED", + "acceptanceSha256": "cd26ee3c8267d84e28d55cbd0753c2112d10b650555d48893f670a6dddd564ee", + "modelEvidence": "SESSION-OBSERVED", + "modelSource": "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", + "state": "judged", + "evidence": "OBSERVED", + "note": "Different tests; both are accepted and both kill 4 of 4 mutants of tick. The TRI arm also found a compiler defect: an assignment to a module var inside a test body is lowered as a new local, so the generated Zig fails with a shadowing error." + }, + { + "id": "t27-4613-validate-trits", + "name": "validate_trits missing test (re-filed 4328)", + "issue": { + "repo": "gHashTag/t27", + "number": 4613, + "url": "https://github.com/gHashTag/t27/issues/4613", + "updatedAt": "2026-09-23T09:31:51Z" + }, + "baseSha": "afe2186cfa9572c8c2d519506aa85a8c9447eae5", + "executorModel": "Claude Code general-purpose subagent; one configuration for both arms, launched together; the exact model id is not written to this repository", + "prompt": "Resolve gHashTag/t27 issue 4613 at the pinned base, working only in the given worktree. Read the issue text (public/queen/runs/campaign-20260926/issues/4613.md). Decision layer: NONE for bee-baseline (no t27c; do not search for one) or TRI for bee-tri (t27c 0.4.0 and zig 0.16.0 on PATH; read-only subcommands only; t27c never writes the patch). Edit only the Boundary file; no network, no commit or push, no toolchain builds. Reply with PATCH, CHECKS and UNVERIFIED.", + "promptSha256": "a444d68feffff727cea821062fdaa1ff836c2a565858a29715de8f27bb4b5a25", + "toolPolicy": "Local read, edit and run tools inside one isolated worktree; python3 tools/dupe_scan.py allowed; no network; no commit or push; no compiler or toolchain builds; edit only the Boundary file. The decision layer is the only difference between arms.", + "toolPolicySha256": "803db5972a1ba27b9139c51f5790e6f477ff1065b36fbc06331a2b166d6ea41a", + "budget": "one agent turn per arm; no token cap; tokens, tool calls and elapsed time recorded from the runtime", + "acceptance": "t27c coverage specs/base/ternary_encoding.t27 reports Untested: 0; function count remains 13; test count is at least 11; t27c spec-status reports IMPLEMENTED; the issue has no test-report criterion", + "acceptanceSha256": "a9378e97de7a7765226678447f2da1fc8b63c0b70531f8c1def9e6c82c52d07b", + "modelEvidence": "SESSION-OBSERVED", + "modelSource": "campaign session 2026-09-26: both arms launched in one message from one session with identical agent settings; the runtime reported the model to that session, and this repository does not record it", + "state": "judged", + "evidence": "OBSERVED", + "note": "Both arms pass the four criteria, but the spec is BLOCKED at the base and this issue has no test-report criterion, so the new test cannot run in the repository. In a review copy with the four pre-existing blockers removed, both tests pass and kill 2 of 4 mutants; the other 2 are caught at compile time by an existing invariant. The TRI arm traced the blocker to balanced-trit encoders paired with unipolar decoders." } ], "runs": [ @@ -203,6 +300,90 @@ export const QUEEN_WARS = { "patchSha256": "11dce31c83d80148e31fdf8b355f5ee11ed436a1a261188296a4a95d01d8a267", "artifactUrl": null, "note": "The Bee added one non-vacuous validate_trits test with four cases, preserving 13 functions and raising the textual test count from 10 to 11. Current t27c v0.2.0 rejected unchanged base syntax at line 37 and reported NOPARSE and BLOCKED. The exact-base compiler accepted the file but dropped all 13 bodies, so executable acceptance is not claimed." + }, + { + "id": "t27-4614-bee-baseline-20260926T185257Z", + "experimentId": "t27-4614-count-buttons", + "configId": "bee-baseline", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:54:00Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "6561a0821694e2a4afa55ddcfe9c4d960223c77487737f0d7e636c713e5f3950", + "patchSha256": "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-baseline-20260926T185257Z", + "note": "Added test count_buttons_is_4 without a compiler; every criterion passed when judged." + }, + { + "id": "t27-4614-bee-tri-20260926T185257Z", + "experimentId": "t27-4614-count-buttons", + "configId": "bee-tri", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:54:02Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "a4db3eb859fb458f0ba512b119ac41c4e350d2d111597927d5f032b80e9ca096", + "patchSha256": "6e3b7f1a1d5ea7edef77875ac4313077f9ae65a794b2c8b39dff08474b5ef0ed", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4614-bee-tri-20260926T185257Z", + "note": "Same patch as the baseline, byte for byte; also ran a mutation check of its own before reporting." + }, + { + "id": "t27-4695-bee-baseline-20260926T185257Z", + "experimentId": "t27-4695-tick", + "configId": "bee-baseline", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:55:26Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "444d27dc87851aaaf5ba7a642df4f6838bbcba043ab99c08dfc47baea5318f24", + "patchSha256": "cf7115ad1f649cc302646c3713488f21d810c47c3f01e1d01a7d72021a889f6b", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-baseline-20260926T185257Z", + "note": "Added test_tick_advances_one_cycle after reset(), argued from the compiler source; every criterion passed when judged." + }, + { + "id": "t27-4695-bee-tri-20260926T185257Z", + "experimentId": "t27-4695-tick", + "configId": "bee-tri", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:55:26Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "4a8ea205b62bede5c36157fe7dc3c7536966c0f6580398f1ce104aef3dd55dc3", + "patchSha256": "aa0443796cf2e7a0ca10dd1b5f885883e6ecaff97d5be77d50a994983e81b7ce", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4695-bee-tri-20260926T185257Z", + "note": "Added test_tick relative to the entry state; found that assigning a module var in a test body generates a shadowing local." + }, + { + "id": "t27-4613-bee-baseline-20260926T185257Z", + "experimentId": "t27-4613-validate-trits", + "configId": "bee-baseline", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:57:10Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "25689dbaf0221824f23f6bcb7ec0f8556fec2cf40832ab226c68333d817b3805", + "patchSha256": "6f44011b4ff970ffd2536c18baf755b64ea69f01539bde11243bcc1fcc743e96", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-baseline-20260926T185257Z", + "note": "Added validate_trits_check with five asserts; the four criteria pass, and test-report stays BLOCKED as at the base." + }, + { + "id": "t27-4613-bee-tri-20260926T185257Z", + "experimentId": "t27-4613-validate-trits", + "configId": "bee-tri", + "startedAt": "2026-09-26T18:52:57Z", + "finishedAt": "2026-09-26T18:56:37Z", + "state": "passed", + "evidence": "OBSERVED", + "verdict": "accepted", + "logSha256": "48716f028e82cc6bc95fe51e3c46de1803a47f2480292ab76c9356d86337dd7a", + "patchSha256": "6c3e1562d398c7e59aacb44b09216463456f295354163590154d42279ded636a", + "artifactUrl": "https://github.com/gHashTag/trinity/tree/7d3a47abe2809bf3864ef2473bde7146ff9af931/apps/website/public/queen/runs/t27-4613-bee-tri-20260926T185257Z", + "note": "Added validate_trits_check with six asserts; traced the pre-existing BLOCKED to mismatched trit encoders and decoders." } ], "measurements": [ @@ -245,6 +426,390 @@ export const QUEEN_WARS = { "unit": "lines", "evidence": "SESSION-OBSERVED", "source": "session-observed git diff --numstat at pinned worktree: 9 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 is recorded in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "elapsed", + "value": "62703", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "tool-calls", + "value": "10", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "patch-lines", + "value": "4", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "total-tokens", + "value": "76852", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "judge-calls", + "value": "0", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4614-bee-baseline-20260926T185257Z", + "key": "mutants-killed", + "value": "3", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "elapsed", + "value": "64182", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "tool-calls", + "value": "14", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "patch-lines", + "value": "4", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 4 insertions and 0 deletions in specs/boards/arty_a7.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "total-tokens", + "value": "77744", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "judge-calls", + "value": "8", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4614-bee-tri-20260926T185257Z", + "key": "mutants-killed", + "value": "3", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 3 of 3 fixed mutants of count_buttons fail the arm's new test" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "elapsed", + "value": "148105", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "tool-calls", + "value": "30", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "patch-lines", + "value": "16", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "total-tokens", + "value": "100044", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "judge-calls", + "value": "0", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4695-bee-baseline-20260926T185257Z", + "key": "mutants-killed", + "value": "4", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "elapsed", + "value": "148367", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "tool-calls", + "value": "27", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "patch-lines", + "value": "16", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 16 insertions and 0 deletions in specs/fpga/testbench/simulator_tb.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "total-tokens", + "value": "98507", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "judge-calls", + "value": "9", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4695-bee-tri-20260926T185257Z", + "key": "mutants-killed", + "value": "4", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 4 of 4 fixed mutants of tick fail the arm's new test" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "elapsed", + "value": "252696", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "tool-calls", + "value": "38", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "patch-lines", + "value": "12", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 12 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "total-tokens", + "value": "128055", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "judge-calls", + "value": "0", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4613-bee-baseline-20260926T185257Z", + "key": "mutants-killed", + "value": "2", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "acceptance", + "value": "PASSED", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "judge transcript acceptance.txt (sha256 in RUN_LOG_SHAS): the issue's own acceptance commands run with t27c 0.4.0 in the arm's worktree" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "queen-verdict", + "value": "accepted", + "unit": "verdict", + "evidence": "OBSERVED", + "source": "Queen review 2026-09-26 of the acceptance transcript, the patch and the fixed-mutant review in mutants.txt" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "elapsed", + "value": "219864", + "unit": "ms", + "evidence": "OBSERVED", + "source": "runtime-reported duration of the arm's agent turn; the start and finish timestamps are the launch record (within 40 s) plus this duration" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "tool-calls", + "value": "27", + "unit": "count", + "evidence": "OBSERVED", + "source": "runtime-reported tool-use count of the arm's agent turn" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "patch-lines", + "value": "18", + "unit": "lines", + "evidence": "OBSERVED", + "source": "git diff --numstat in the arm's worktree: 18 insertions and 0 deletions in specs/base/ternary_encoding.t27; patch sha256 in RUN_PATCH_SHAS" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "total-tokens", + "value": "113494", + "unit": "tokens", + "evidence": "OBSERVED", + "source": "runtime-reported token total of the arm's agent turn; input and output are not reported separately" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "judge-calls", + "value": "17", + "unit": "count", + "evidence": "OBSERVED", + "source": "shell commands in the arm's transcript that invoked t27c" + }, + { + "runId": "t27-4613-bee-tri-20260926T185257Z", + "key": "mutants-killed", + "value": "2", + "unit": "count", + "evidence": "OBSERVED", + "source": "mutants.txt: 2 of 4 fixed mutants of validate_trits fail the arm's new test in a review copy with the four pre-existing blockers removed; the other 2 mutants do not compile" } ] } as const diff --git a/apps/website/src/lib/roadmapGame.ts b/apps/website/src/lib/roadmapGame.ts new file mode 100644 index 0000000000..6606a7fb91 --- /dev/null +++ b/apps/website/src/lib/roadmapGame.ts @@ -0,0 +1,137 @@ +// The rules of LEVEL II, as pure functions, so a contract can hold them. +// +// Everything here is computed from public facts - GitHub's issue search, the +// Queen's public board and the roadmap's goals.json - and nothing is stored. +// The page's own score (honey) and its own rules (the raid, when a boss fight +// opens) are named as the board's, never as the Queen's XP, which only the +// Queen's leaderboard reports. + +export interface RuleGoal { + stage: number + locked?: { en: string; ru: string } +} + +export interface RuleTask { + stage: number | null + units: number + state: string + /** ISO time the issue was closed, for built cells. */ + closedAt?: string | null +} + +const DAY_MS = 86_400_000 + +/** Whole UTC days since the epoch: the raid changes at UTC midnight. */ +export function utcDay(nowMs: number): number { + return Math.floor(nowMs / DAY_MS) +} + +/** Milliseconds until the next UTC midnight, when the raid moves. */ +export function msToNextDay(nowMs: number): number { + return DAY_MS - (nowMs % DAY_MS) +} + +/** + * The stages a raid can fall on: stages 1-7 with nothing locking them. + * With today's goals.json that is [1, 2, 5, 6, 7], and gHashTag/t27 + * tools/queen/feed_roadmap.py holds the same list as RAID_STAGES, so the + * feeder and this page name the same sector on the same UTC day. Unlocking a + * stage in goals.json means changing that list too. + */ +export function raidStages(goals: RuleGoal[]): number[] { + return goals + .filter((g) => g.stage >= 1 && g.stage <= 7 && !g.locked) + .map((g) => g.stage) + .sort((a, b) => a - b) +} + +/** Today's raid sector: the eligible stages in turn, one per UTC day. */ +export function raidOf(goals: RuleGoal[], day: number): number | null { + const stages = raidStages(goals) + return stages.length ? stages[((day % stages.length) + stages.length) % stages.length] : null +} + +/** + * Honey: the functions ported in built cells. A cell of the raid sector that + * was closed today counts double. This is the board's own score. + */ +export function honeyOf(tasks: RuleTask[], raid: number | null, day: number): number { + let honey = 0 + for (const t of tasks) { + if (t.state !== 'built' && t.state !== 'cracked') continue + const today = t.closedAt ? utcDay(Date.parse(t.closedAt)) === day : false + honey += t.units * (raid !== null && t.stage === raid && today ? 2 : 1) + } + return honey +} + +/** The share of the comb below the bosses (stages 1-7) that is built, 0..1. */ +export function builtShareBelow(tasks: RuleTask[]): number { + const below = tasks.filter((t) => t.stage !== null && t.stage <= 7) + if (below.length === 0) return 0 + return below.filter((t) => t.state === 'built' || t.state === 'cracked').length / below.length +} + +/** A boss fight opens when this share of the comb below it is built. */ +export const BOSS_OPENS_AT = 0.5 + +/** The `.t27` a port task creates, read from its title: `... to specs/port/.t27`. */ +export function targetOf(title: string): string | null { + const m = /\bto (specs\/port\/\S+?\.t27)\b/.exec(title) + return m ? m[1] : null +} + +/** + * Built files with an open defect against them. A title that names a + * `specs/port/...t27` and is not itself a port task is a report that the + * file, once built, is wrong now - the queen-doctor and the feeders file them + * that way. Each such file is a cracked cell until its defect closes. + */ +export function crackedTargets(openTitles: string[]): Set { + const out = new Set() + for (const title of openTitles) { + if (/^Port /.test(title)) continue + for (const m of title.matchAll(/specs\/port\/[\w./-]+?\.t27/g)) out.add(m[0]) + } + return out +} + +/** + * Where the Queen's round is, 0 at its start and 1 at its end, from the + * board's own pulse. Null when the board gave no usable pulse. + */ +export function pulsePhase(lastRoundAt: string | null, roundSeconds: number, nowMs: number): number | null { + if (!lastRoundAt || !(roundSeconds > 0)) return null + const at = Date.parse(lastRoundAt) + if (!Number.isFinite(at)) return null + const elapsed = Math.max(0, (nowMs - at) / 1000) + return (elapsed % roundSeconds) / roundSeconds +} + +export const EXT_LANG: Record = { + py: 'Python', sh: 'Shell', ts: 'TypeScript', mts: 'TypeScript', js: 'JavaScript', mjs: 'JavaScript', + go: 'Go', rs: 'Rust', zig: 'Zig', c: 'C', v: 'Verilog', sv: 'Verilog', gleam: 'Gleam', +} + +/** `Port [part N of M of ][owner/repo:]path (Lang, N functions) to ...` */ +export function parsePortTitle(title: string): { repo: string; path: string; lang: string; units: number } | null { + const m = /^Port (?:part \d+ of \d+ of )?(?:([\w.-]+\/[\w.-]+):)?(\S+)(?: \(([\w+#]+), (\d+) (?:functions?|modules?)\))?/.exec(title) + if (!m) return null + const path = m[2] + const ext = path.includes('.') ? path.slice(path.lastIndexOf('.') + 1) : '' + return { + repo: m[1] ?? 'gHashTag/t27', + path, + lang: m[3] ?? EXT_LANG[ext] ?? '', + // A title that does not say is read as middling, not as quick. + units: m[4] ? Number(m[4]) : 4, + } +} + +/** Rows of an apex-down pyramid, bottom (the point) first: row r holds r + 1 cells. */ +export function rowsFor(cells: number): number { + let rows = 1 + while ((rows * (rows + 1)) / 2 < cells) rows += 1 + return Math.min(Math.max(rows, 9), 16) +} +