From 0770abb994bfa78ab1310dff60d273da67392871 Mon Sep 17 00:00:00 2001 From: Dmitriy Vasilev Date: Sat, 26 Sep 2026 19:26:54 +0700 Subject: [PATCH] feat(wars): the arena's decision layer is TRI; JEV is only a comparison arm Renames the variable factor from the external jev-decision-layer to tri-decision-layer: the judge is the t27c compiler itself (OBSERVED, the spec is its own source), so an arm is accepted only when the compiler and the acceptance gates accept its patch. The Bee + JEV arm stays in the ledger purely as a comparison baseline. Co-Authored-By: Crush:glm-4.5 --- apps/website/public/queen/wars.json | 10 +++++----- apps/website/public/queen/wars.t27 | 12 ++++++------ apps/website/qa/queen-wars-contract.mjs | 10 +++++----- apps/website/scripts/queen-wars-from-spec.mjs | 10 +++++----- apps/website/scripts/queen-wars-from-spec.test.mjs | 2 +- apps/website/specs/queen/wars.t27 | 12 ++++++------ apps/website/src/components/QueenWars.tsx | 12 ++++++------ apps/website/src/lib/queenModules.ts | 4 ++-- apps/website/src/lib/queenWars.generated.ts | 12 ++++++------ 9 files changed, 42 insertions(+), 42 deletions(-) diff --git a/apps/website/public/queen/wars.json b/apps/website/public/queen/wars.json index 71ec97fde5..308cf1bf46 100644 --- a/apps/website/public/queen/wars.json +++ b/apps/website/public/queen/wars.json @@ -2,7 +2,7 @@ "source": { "spec": "specs/queen/wars.t27", "publicSpec": "public/queen/wars.t27", - "sha256": "b7c1f22ea8361a092c083d0a2681477af2d7ec08bd35802dfd25289f0e784924", + "sha256": "fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8", "schemaVersion": 2 }, "name": "Queen WARS real-task agent arena", @@ -43,7 +43,7 @@ }, "protocol": { "realGitHubTasksOnly": true, - "variableFactor": "jev-decision-layer", + "variableFactor": "tri-decision-layer", "controlledFactors": [ "issue-url", "base-sha", @@ -58,9 +58,9 @@ "reviewPolicy": "Queen reviews the diff and exact evidence after automated gates", "winnerPolicy": "merge at most one accepted patch; preserve every arm in the append-only ledger", "comparisonValidityPolicy": "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner", - "jevRole": "structured decision layer that ranks typed choices; never a chat model and never a code generator", - "jevRoleEvidence": "SOURCE-CLAIM", - "jevRoleSource": "https://docs.typesafe.ai/introduction/coding-agents", + "triRole": "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch", + "triRoleEvidence": "OBSERVED", + "triRoleSource": "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27", "emptyMetricMeansUnknown": true }, "metricCatalog": [ diff --git a/apps/website/public/queen/wars.t27 b/apps/website/public/queen/wars.t27 index 503ea06e2a..a49983c9fb 100644 --- a/apps/website/public/queen/wars.t27 +++ b/apps/website/public/queen/wars.t27 @@ -25,16 +25,16 @@ pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; // Fairness contract. A result that breaks one controlled factor is not an A/B result. pub const REAL_GITHUB_TASKS_ONLY : bool = true; -pub const VARIABLE_FACTOR : str = "jev-decision-layer"; +pub const VARIABLE_FACTOR : str = "tri-decision-layer"; pub const CONTROLLED_FACTORS : [7]str = ["issue-url", "base-sha", "executor-model", "prompt", "tool-policy", "budget", "acceptance"]; pub const ISOLATION : str = "one clean git worktree per arm, both pinned to the same base commit"; pub const ACCEPTANCE_POLICY : str = "the issue acceptance commands plus repository gates decide pass or fail"; pub const REVIEW_POLICY : str = "Queen reviews the diff and exact evidence after automated gates"; pub const WINNER_POLICY : str = "merge at most one accepted patch; preserve every arm in the append-only ledger"; pub const COMPARISON_VALIDITY_POLICY : str = "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner"; -pub const JEV_ROLE : str = "structured decision layer that ranks typed choices; never a chat model and never a code generator"; -pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM"; -pub const JEV_ROLE_SOURCE : str = "https://docs.typesafe.ai/introduction/coding-agents"; +pub const TRI_ROLE : str = "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch"; +pub const TRI_ROLE_EVIDENCE : str = "OBSERVED"; +pub const TRI_ROLE_SOURCE : str = "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27"; // Measurement vocabulary. Values are strings so an exact native unit is preserved; // absent rows mean unknown. A numeric zero is a measured zero only when a source exists. @@ -102,11 +102,11 @@ pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree co test protocol_has_one_variable { assert REAL_GITHUB_TASKS_ONLY == true; - assert VARIABLE_FACTOR == "jev-decision-layer"; + assert VARIABLE_FACTOR == "tri-decision-layer"; assert CONTROLLED_FACTORS[0] == "issue-url"; assert CONTROLLED_FACTORS[6] == "acceptance"; assert EMPTY_METRIC_MEANS_UNKNOWN == true; - assert JEV_ROLE_EVIDENCE == "SOURCE-CLAIM"; + assert TRI_ROLE_EVIDENCE == "OBSERVED"; } test arena_names_all_four_lanes { diff --git a/apps/website/qa/queen-wars-contract.mjs b/apps/website/qa/queen-wars-contract.mjs index 399824201c..a679eb386c 100644 --- a/apps/website/qa/queen-wars-contract.mjs +++ b/apps/website/qa/queen-wars-contract.mjs @@ -21,8 +21,8 @@ for (const file of [SPEC, GENERATOR, TS, JSON_OUT, PUBLIC_SPEC, COMPONENT, CSS]) const spec = read(SPEC) assert.doesNotMatch(spec, /[^\x00-\x7f]/, 'the WARS .t27 source must remain ASCII/L3') assert.match(spec, /pub const REAL_GITHUB_TASKS_ONLY : bool = true;/) -assert.match(spec, /pub const VARIABLE_FACTOR : str = "jev-decision-layer";/) -assert.match(spec, /pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM";/) +assert.match(spec, /pub const VARIABLE_FACTOR : str = "tri-decision-layer";/) +assert.match(spec, /pub const TRI_ROLE_EVIDENCE : str = "OBSERVED";/) assert.match(spec, /https:\/\/docs\.typesafe\.ai\/introduction\/coding-agents/) assert.match(spec, /https:\/\/github\.com\/gHashTag\/t27\/issues\/4328/) assert.match(spec, /current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth/) @@ -34,9 +34,9 @@ const generated = JSON.parse(read(JSON_OUT)) assert.equal(generated.source.spec, SPEC) assert.match(generated.source.sha256, /^[0-9a-f]{64}$/) assert.deepEqual(generated.configurations.map((item) => item.id), ['bee-baseline', 'bee-jev', 'igla-coder', 'igla-race']) -assert.equal(generated.protocol.variableFactor, 'jev-decision-layer') -assert.equal(generated.protocol.jevRoleEvidence, 'SOURCE-CLAIM') -assert.match(generated.protocol.jevRoleSource, /^https:\/\/docs\.typesafe\.ai\//) +assert.equal(generated.protocol.variableFactor, 'tri-decision-layer') +assert.equal(generated.protocol.triRoleEvidence, 'OBSERVED') +assert.match(generated.protocol.triRoleSource, /^https:\/\/t27\.ai\//) assert.equal(generated.configurations[1].evidence, 'SOURCE-CLAIM') assert.equal(generated.configurations[1].stateEvidence, 'OBSERVED') assert.equal(generated.experiments[0].issue.url, 'https://github.com/gHashTag/t27/issues/4328') diff --git a/apps/website/scripts/queen-wars-from-spec.mjs b/apps/website/scripts/queen-wars-from-spec.mjs index 98a8ef1629..c35d0ddfea 100644 --- a/apps/website/scripts/queen-wars-from-spec.mjs +++ b/apps/website/scripts/queen-wars-from-spec.mjs @@ -28,7 +28,7 @@ const REQUIRED = { EVIDENCE_LEVELS: 'arr', CONFIG_STATES: 'arr', EXPERIMENT_STATES: 'arr', RUN_STATES: 'arr', VERDICTS: 'arr', EMPTY_METRIC_MEANS_UNKNOWN: 'bool', REAL_GITHUB_TASKS_ONLY: 'bool', VARIABLE_FACTOR: 'str', CONTROLLED_FACTORS: 'arr', ISOLATION: 'str', ACCEPTANCE_POLICY: 'str', REVIEW_POLICY: 'str', WINNER_POLICY: 'str', COMPARISON_VALIDITY_POLICY: 'str', - JEV_ROLE: 'str', JEV_ROLE_EVIDENCE: 'str', JEV_ROLE_SOURCE: 'str', + TRI_ROLE: 'str', TRI_ROLE_EVIDENCE: 'str', TRI_ROLE_SOURCE: 'str', METRIC_COUNT: 'u8', METRIC_KEYS: 'arr', METRIC_UNITS: 'arr', CONFIG_COUNT: 'u8', CONFIG_IDS: 'arr', CONFIG_NAMES: 'arr', CONFIG_KINDS: 'arr', CONFIG_STATES_BY_ID: 'arr', CONFIG_EVIDENCE: 'arr', CONFIG_SOURCES: 'arr', CONFIG_NOTES: 'arr', CONFIG_STATE_EVIDENCE: 'arr', @@ -99,9 +99,9 @@ export function semanticProblems(f, file = WARS_SPEC) { } if (f.REAL_GITHUB_TASKS_ONLY !== true) p.push(`${file}: REAL_GITHUB_TASKS_ONLY must be true`) if (f.EMPTY_METRIC_MEANS_UNKNOWN !== true) p.push(`${file}: EMPTY_METRIC_MEANS_UNKNOWN must be true`) - if (f.VARIABLE_FACTOR !== 'jev-decision-layer') p.push(`${file}: VARIABLE_FACTOR must be jev-decision-layer`) - if (!oneOf(f.JEV_ROLE_EVIDENCE, f.EVIDENCE_LEVELS)) p.push(`${file}: JEV_ROLE_EVIDENCE is unknown`) - if (!f.JEV_ROLE_SOURCE.trim()) p.push(`${file}: JEV_ROLE_SOURCE is empty`) + if (f.VARIABLE_FACTOR !== 'tri-decision-layer') p.push(`${file}: VARIABLE_FACTOR must be tri-decision-layer`) + if (!oneOf(f.TRI_ROLE_EVIDENCE, f.EVIDENCE_LEVELS)) p.push(`${file}: TRI_ROLE_EVIDENCE is unknown`) + if (!f.TRI_ROLE_SOURCE.trim()) p.push(`${file}: TRI_ROLE_SOURCE is empty`) for (const [countName, arrays] of Object.entries(PARALLEL)) { for (const name of arrays) if (f[name].length !== f[countName]) p.push(`${file}: ${name}.length ${f[name].length} != ${countName} ${f[countName]}`) } @@ -252,7 +252,7 @@ export function arenaOf(f, specSha) { realGitHubTasksOnly: f.REAL_GITHUB_TASKS_ONLY, variableFactor: f.VARIABLE_FACTOR, controlledFactors: f.CONTROLLED_FACTORS, isolation: f.ISOLATION, acceptancePolicy: f.ACCEPTANCE_POLICY, reviewPolicy: f.REVIEW_POLICY, winnerPolicy: f.WINNER_POLICY, comparisonValidityPolicy: f.COMPARISON_VALIDITY_POLICY, - jevRole: f.JEV_ROLE, jevRoleEvidence: f.JEV_ROLE_EVIDENCE, jevRoleSource: f.JEV_ROLE_SOURCE, + triRole: f.TRI_ROLE, triRoleEvidence: f.TRI_ROLE_EVIDENCE, triRoleSource: f.TRI_ROLE_SOURCE, emptyMetricMeansUnknown: f.EMPTY_METRIC_MEANS_UNKNOWN, }, metricCatalog: rows(f.METRIC_COUNT, (i) => ({ key: f.METRIC_KEYS[i], unit: f.METRIC_UNITS[i] })), diff --git a/apps/website/scripts/queen-wars-from-spec.test.mjs b/apps/website/scripts/queen-wars-from-spec.test.mjs index 4df558730e..3580916729 100644 --- a/apps/website/scripts/queen-wars-from-spec.test.mjs +++ b/apps/website/scripts/queen-wars-from-spec.test.mjs @@ -60,7 +60,7 @@ test('the WARS source compiles, evaluates its tests and renders all projections' assert.match(out.arena.configurations[1].source, /^https:\/\/docs\.typesafe\.ai\//) assert.equal(out.arena.experiments[0].issue.number, 4328) assert.equal(out.arena.experiments[0].modelEvidence, 'UNKNOWN') - assert.equal(out.arena.protocol.jevRoleEvidence, 'SOURCE-CLAIM') + assert.equal(out.arena.protocol.triRoleEvidence, 'OBSERVED') assert.equal(out.arena.runs.length, 1) assert.equal(out.arena.measurements.length, 5) }) diff --git a/apps/website/specs/queen/wars.t27 b/apps/website/specs/queen/wars.t27 index 503ea06e2a..a49983c9fb 100644 --- a/apps/website/specs/queen/wars.t27 +++ b/apps/website/specs/queen/wars.t27 @@ -25,16 +25,16 @@ pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; // Fairness contract. A result that breaks one controlled factor is not an A/B result. pub const REAL_GITHUB_TASKS_ONLY : bool = true; -pub const VARIABLE_FACTOR : str = "jev-decision-layer"; +pub const VARIABLE_FACTOR : str = "tri-decision-layer"; pub const CONTROLLED_FACTORS : [7]str = ["issue-url", "base-sha", "executor-model", "prompt", "tool-policy", "budget", "acceptance"]; pub const ISOLATION : str = "one clean git worktree per arm, both pinned to the same base commit"; pub const ACCEPTANCE_POLICY : str = "the issue acceptance commands plus repository gates decide pass or fail"; pub const REVIEW_POLICY : str = "Queen reviews the diff and exact evidence after automated gates"; pub const WINNER_POLICY : str = "merge at most one accepted patch; preserve every arm in the append-only ledger"; pub const COMPARISON_VALIDITY_POLICY : str = "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner"; -pub const JEV_ROLE : str = "structured decision layer that ranks typed choices; never a chat model and never a code generator"; -pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM"; -pub const JEV_ROLE_SOURCE : str = "https://docs.typesafe.ai/introduction/coding-agents"; +pub const TRI_ROLE : str = "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch"; +pub const TRI_ROLE_EVIDENCE : str = "OBSERVED"; +pub const TRI_ROLE_SOURCE : str = "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27"; // Measurement vocabulary. Values are strings so an exact native unit is preserved; // absent rows mean unknown. A numeric zero is a measured zero only when a source exists. @@ -102,11 +102,11 @@ pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree co test protocol_has_one_variable { assert REAL_GITHUB_TASKS_ONLY == true; - assert VARIABLE_FACTOR == "jev-decision-layer"; + assert VARIABLE_FACTOR == "tri-decision-layer"; assert CONTROLLED_FACTORS[0] == "issue-url"; assert CONTROLLED_FACTORS[6] == "acceptance"; assert EMPTY_METRIC_MEANS_UNKNOWN == true; - assert JEV_ROLE_EVIDENCE == "SOURCE-CLAIM"; + assert TRI_ROLE_EVIDENCE == "OBSERVED"; } test arena_names_all_four_lanes { diff --git a/apps/website/src/components/QueenWars.tsx b/apps/website/src/components/QueenWars.tsx index aae24f8ef8..f783aa83a4 100644 --- a/apps/website/src/components/QueenWars.tsx +++ b/apps/website/src/components/QueenWars.tsx @@ -36,7 +36,7 @@ const COPY = { factor: 'variable', fixed: 'held constant', comparisonRule: 'comparison validity', - jevRole: 'JEV role', + triRole: 'TRI role (the judge)', task: 'REAL TASK', experimentPicker: 'EXPERIMENT', base: 'base', @@ -74,7 +74,7 @@ const COPY = { factor: 'переменная', fixed: 'зафиксировано', comparisonRule: 'валидность сравнения', - jevRole: 'роль JEV', + triRole: 'Роль TRI (судья)', task: 'РЕАЛЬНАЯ ЗАДАЧА', experimentPicker: 'ЭКСПЕРИМЕНТ', base: 'база', @@ -188,11 +188,11 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {
{QUEEN_WARS.protocol.comparisonValidityPolicy}
-
{c.jevRole}
+
{c.triRole}
- {QUEEN_WARS.protocol.jevRole} - - {c.evidenceSource} + {QUEEN_WARS.protocol.triRole} + + {c.evidenceSource}
diff --git a/apps/website/src/lib/queenModules.ts b/apps/website/src/lib/queenModules.ts index 3d378efb1b..5b6d0e97fa 100644 --- a/apps/website/src/lib/queenModules.ts +++ b/apps/website/src/lib/queenModules.ts @@ -320,13 +320,13 @@ export const MODULES = [ en: { name: 'WARS', hint: 'Real-task agent benchmarks generated from one .t27 ledger (key x)', - body: 'The controlled arena for Bees, JEV-assisted decisions, IGLA CODER and IGLA RACE. Every experiment pins a real GitHub issue, base commit, prompt, tools, budget and acceptance gates in specs/queen/wars.t27; missing runs remain unknown rather than becoming zero, and the interface is only a generated projection of that ledger.', + body: 'The controlled arena for Bees, TRI-assisted decisions (the compiler judge; JEV remains only as a comparison arm), IGLA CODER and IGLA RACE. Every experiment pins a real GitHub issue, base commit, prompt, tools, budget and acceptance gates in specs/queen/wars.t27; missing runs remain unknown rather than becoming zero, and the interface is only a generated projection of that ledger.', play: 'Where an agent configuration earns its place. Compare witnessed work under equal conditions, keep one accepted patch, and carry every losing or blocked arm forward as evidence rather than erasing it.', }, ru: { name: 'ВОЙНЫ', hint: 'Бенчмарки агентов на реальных задачах из единого журнала .t27 (клавиша x)', - body: 'Контролируемая арена для Bees, решений с JEV, IGLA CODER и IGLA RACE. Каждый эксперимент закрепляет реальную GitHub issue, базовый commit, prompt, инструменты, бюджет и ворота приёмки в specs/queen/wars.t27; отсутствующие запуски остаются неизвестными, а не превращаются в нули, а интерфейс служит только порождённой проекцией этого журнала.', + body: 'Контролируемая арена для Bees, решений с TRI (судья-компилятор; JEV остался только сравнительным плечом), IGLA CODER и IGLA RACE. Каждый эксперимент закрепляет реальную GitHub issue, базовый commit, prompt, инструменты, бюджет и ворота приёмки в specs/queen/wars.t27; отсутствующие запуски остаются неизвестными, а не превращаются в нули, а интерфейс служит только порождённой проекцией этого журнала.', play: 'Место, где конфигурация агента заслуживает своё место. Сравнивайте подтверждённую работу в равных условиях, принимайте только один patch и сохраняйте проигравшие или заблокированные руки как свидетельство.', }, }, diff --git a/apps/website/src/lib/queenWars.generated.ts b/apps/website/src/lib/queenWars.generated.ts index b43e3870cf..e15d9bff1e 100644 --- a/apps/website/src/lib/queenWars.generated.ts +++ b/apps/website/src/lib/queenWars.generated.ts @@ -1,12 +1,12 @@ // GENERATED by scripts/queen-wars-from-spec.mjs from specs/queen/wars.t27 -// spec sha256 b7c1f22ea8361a092c083d0a2681477af2d7ec08bd35802dfd25289f0e784924 +// spec sha256 fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8 // Do not edit: change the .t27 source and regenerate. export const QUEEN_WARS = { "source": { "spec": "specs/queen/wars.t27", "publicSpec": "public/queen/wars.t27", - "sha256": "b7c1f22ea8361a092c083d0a2681477af2d7ec08bd35802dfd25289f0e784924", + "sha256": "fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8", "schemaVersion": 2 }, "name": "Queen WARS real-task agent arena", @@ -47,7 +47,7 @@ export const QUEEN_WARS = { }, "protocol": { "realGitHubTasksOnly": true, - "variableFactor": "jev-decision-layer", + "variableFactor": "tri-decision-layer", "controlledFactors": [ "issue-url", "base-sha", @@ -62,9 +62,9 @@ export const QUEEN_WARS = { "reviewPolicy": "Queen reviews the diff and exact evidence after automated gates", "winnerPolicy": "merge at most one accepted patch; preserve every arm in the append-only ledger", "comparisonValidityPolicy": "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner", - "jevRole": "structured decision layer that ranks typed choices; never a chat model and never a code generator", - "jevRoleEvidence": "SOURCE-CLAIM", - "jevRoleSource": "https://docs.typesafe.ai/introduction/coding-agents", + "triRole": "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch", + "triRoleEvidence": "OBSERVED", + "triRoleSource": "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27", "emptyMetricMeansUnknown": true }, "metricCatalog": [