diff --git a/apps/website/public/queen/wars.json b/apps/website/public/queen/wars.json index 71ec97fde5..308cf1bf46 100644 --- a/apps/website/public/queen/wars.json +++ b/apps/website/public/queen/wars.json @@ -2,7 +2,7 @@ "source": { "spec": "specs/queen/wars.t27", "publicSpec": "public/queen/wars.t27", - "sha256": "b7c1f22ea8361a092c083d0a2681477af2d7ec08bd35802dfd25289f0e784924", + "sha256": "fc95edacac2dd884093d4a969cf7b211558a418216981e8300d44c505473c1d8", "schemaVersion": 2 }, "name": "Queen WARS real-task agent arena", @@ -43,7 +43,7 @@ }, "protocol": { "realGitHubTasksOnly": true, - "variableFactor": "jev-decision-layer", + "variableFactor": "tri-decision-layer", "controlledFactors": [ "issue-url", "base-sha", @@ -58,9 +58,9 @@ "reviewPolicy": "Queen reviews the diff and exact evidence after automated gates", "winnerPolicy": "merge at most one accepted patch; preserve every arm in the append-only ledger", "comparisonValidityPolicy": "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner", - "jevRole": "structured decision layer that ranks typed choices; never a chat model and never a code generator", - "jevRoleEvidence": "SOURCE-CLAIM", - "jevRoleSource": "https://docs.typesafe.ai/introduction/coding-agents", + "triRole": "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch", + "triRoleEvidence": "OBSERVED", + "triRoleSource": "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27", "emptyMetricMeansUnknown": true }, "metricCatalog": [ diff --git a/apps/website/public/queen/wars.t27 b/apps/website/public/queen/wars.t27 index 503ea06e2a..a49983c9fb 100644 --- a/apps/website/public/queen/wars.t27 +++ b/apps/website/public/queen/wars.t27 @@ -25,16 +25,16 @@ pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; // Fairness contract. A result that breaks one controlled factor is not an A/B result. pub const REAL_GITHUB_TASKS_ONLY : bool = true; -pub const VARIABLE_FACTOR : str = "jev-decision-layer"; +pub const VARIABLE_FACTOR : str = "tri-decision-layer"; pub const CONTROLLED_FACTORS : [7]str = ["issue-url", "base-sha", "executor-model", "prompt", "tool-policy", "budget", "acceptance"]; pub const ISOLATION : str = "one clean git worktree per arm, both pinned to the same base commit"; pub const ACCEPTANCE_POLICY : str = "the issue acceptance commands plus repository gates decide pass or fail"; pub const REVIEW_POLICY : str = "Queen reviews the diff and exact evidence after automated gates"; pub const WINNER_POLICY : str = "merge at most one accepted patch; preserve every arm in the append-only ledger"; pub const COMPARISON_VALIDITY_POLICY : str = "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner"; -pub const JEV_ROLE : str = "structured decision layer that ranks typed choices; never a chat model and never a code generator"; -pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM"; -pub const JEV_ROLE_SOURCE : str = "https://docs.typesafe.ai/introduction/coding-agents"; +pub const TRI_ROLE : str = "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch"; +pub const TRI_ROLE_EVIDENCE : str = "OBSERVED"; +pub const TRI_ROLE_SOURCE : str = "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27"; // Measurement vocabulary. Values are strings so an exact native unit is preserved; // absent rows mean unknown. A numeric zero is a measured zero only when a source exists. @@ -102,11 +102,11 @@ pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree co test protocol_has_one_variable { assert REAL_GITHUB_TASKS_ONLY == true; - assert VARIABLE_FACTOR == "jev-decision-layer"; + assert VARIABLE_FACTOR == "tri-decision-layer"; assert CONTROLLED_FACTORS[0] == "issue-url"; assert CONTROLLED_FACTORS[6] == "acceptance"; assert EMPTY_METRIC_MEANS_UNKNOWN == true; - assert JEV_ROLE_EVIDENCE == "SOURCE-CLAIM"; + assert TRI_ROLE_EVIDENCE == "OBSERVED"; } test arena_names_all_four_lanes { diff --git a/apps/website/qa/queen-wars-contract.mjs b/apps/website/qa/queen-wars-contract.mjs index 399824201c..a679eb386c 100644 --- a/apps/website/qa/queen-wars-contract.mjs +++ b/apps/website/qa/queen-wars-contract.mjs @@ -21,8 +21,8 @@ for (const file of [SPEC, GENERATOR, TS, JSON_OUT, PUBLIC_SPEC, COMPONENT, CSS]) const spec = read(SPEC) assert.doesNotMatch(spec, /[^\x00-\x7f]/, 'the WARS .t27 source must remain ASCII/L3') assert.match(spec, /pub const REAL_GITHUB_TASKS_ONLY : bool = true;/) -assert.match(spec, /pub const VARIABLE_FACTOR : str = "jev-decision-layer";/) -assert.match(spec, /pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM";/) +assert.match(spec, /pub const VARIABLE_FACTOR : str = "tri-decision-layer";/) +assert.match(spec, /pub const TRI_ROLE_EVIDENCE : str = "OBSERVED";/) assert.match(spec, /https:\/\/docs\.typesafe\.ai\/introduction\/coding-agents/) assert.match(spec, /https:\/\/github\.com\/gHashTag\/t27\/issues\/4328/) assert.match(spec, /current process environment, GitHub Actions secret names, local Railway IaC, and local wrangler auth/) @@ -34,9 +34,9 @@ const generated = JSON.parse(read(JSON_OUT)) assert.equal(generated.source.spec, SPEC) assert.match(generated.source.sha256, /^[0-9a-f]{64}$/) assert.deepEqual(generated.configurations.map((item) => item.id), ['bee-baseline', 'bee-jev', 'igla-coder', 'igla-race']) -assert.equal(generated.protocol.variableFactor, 'jev-decision-layer') -assert.equal(generated.protocol.jevRoleEvidence, 'SOURCE-CLAIM') -assert.match(generated.protocol.jevRoleSource, /^https:\/\/docs\.typesafe\.ai\//) +assert.equal(generated.protocol.variableFactor, 'tri-decision-layer') +assert.equal(generated.protocol.triRoleEvidence, 'OBSERVED') +assert.match(generated.protocol.triRoleSource, /^https:\/\/t27\.ai\//) assert.equal(generated.configurations[1].evidence, 'SOURCE-CLAIM') assert.equal(generated.configurations[1].stateEvidence, 'OBSERVED') assert.equal(generated.experiments[0].issue.url, 'https://github.com/gHashTag/t27/issues/4328') diff --git a/apps/website/scripts/queen-wars-from-spec.mjs b/apps/website/scripts/queen-wars-from-spec.mjs index 98a8ef1629..c35d0ddfea 100644 --- a/apps/website/scripts/queen-wars-from-spec.mjs +++ b/apps/website/scripts/queen-wars-from-spec.mjs @@ -28,7 +28,7 @@ const REQUIRED = { EVIDENCE_LEVELS: 'arr', CONFIG_STATES: 'arr', EXPERIMENT_STATES: 'arr', RUN_STATES: 'arr', VERDICTS: 'arr', EMPTY_METRIC_MEANS_UNKNOWN: 'bool', REAL_GITHUB_TASKS_ONLY: 'bool', VARIABLE_FACTOR: 'str', CONTROLLED_FACTORS: 'arr', ISOLATION: 'str', ACCEPTANCE_POLICY: 'str', REVIEW_POLICY: 'str', WINNER_POLICY: 'str', COMPARISON_VALIDITY_POLICY: 'str', - JEV_ROLE: 'str', JEV_ROLE_EVIDENCE: 'str', JEV_ROLE_SOURCE: 'str', + TRI_ROLE: 'str', TRI_ROLE_EVIDENCE: 'str', TRI_ROLE_SOURCE: 'str', METRIC_COUNT: 'u8', METRIC_KEYS: 'arr', METRIC_UNITS: 'arr', CONFIG_COUNT: 'u8', CONFIG_IDS: 'arr', CONFIG_NAMES: 'arr', CONFIG_KINDS: 'arr', CONFIG_STATES_BY_ID: 'arr', CONFIG_EVIDENCE: 'arr', CONFIG_SOURCES: 'arr', CONFIG_NOTES: 'arr', CONFIG_STATE_EVIDENCE: 'arr', @@ -99,9 +99,9 @@ export function semanticProblems(f, file = WARS_SPEC) { } if (f.REAL_GITHUB_TASKS_ONLY !== true) p.push(`${file}: REAL_GITHUB_TASKS_ONLY must be true`) if (f.EMPTY_METRIC_MEANS_UNKNOWN !== true) p.push(`${file}: EMPTY_METRIC_MEANS_UNKNOWN must be true`) - if (f.VARIABLE_FACTOR !== 'jev-decision-layer') p.push(`${file}: VARIABLE_FACTOR must be jev-decision-layer`) - if (!oneOf(f.JEV_ROLE_EVIDENCE, f.EVIDENCE_LEVELS)) p.push(`${file}: JEV_ROLE_EVIDENCE is unknown`) - if (!f.JEV_ROLE_SOURCE.trim()) p.push(`${file}: JEV_ROLE_SOURCE is empty`) + if (f.VARIABLE_FACTOR !== 'tri-decision-layer') p.push(`${file}: VARIABLE_FACTOR must be tri-decision-layer`) + if (!oneOf(f.TRI_ROLE_EVIDENCE, f.EVIDENCE_LEVELS)) p.push(`${file}: TRI_ROLE_EVIDENCE is unknown`) + if (!f.TRI_ROLE_SOURCE.trim()) p.push(`${file}: TRI_ROLE_SOURCE is empty`) for (const [countName, arrays] of Object.entries(PARALLEL)) { for (const name of arrays) if (f[name].length !== f[countName]) p.push(`${file}: ${name}.length ${f[name].length} != ${countName} ${f[countName]}`) } @@ -252,7 +252,7 @@ export function arenaOf(f, specSha) { realGitHubTasksOnly: f.REAL_GITHUB_TASKS_ONLY, variableFactor: f.VARIABLE_FACTOR, controlledFactors: f.CONTROLLED_FACTORS, isolation: f.ISOLATION, acceptancePolicy: f.ACCEPTANCE_POLICY, reviewPolicy: f.REVIEW_POLICY, winnerPolicy: f.WINNER_POLICY, comparisonValidityPolicy: f.COMPARISON_VALIDITY_POLICY, - jevRole: f.JEV_ROLE, jevRoleEvidence: f.JEV_ROLE_EVIDENCE, jevRoleSource: f.JEV_ROLE_SOURCE, + triRole: f.TRI_ROLE, triRoleEvidence: f.TRI_ROLE_EVIDENCE, triRoleSource: f.TRI_ROLE_SOURCE, emptyMetricMeansUnknown: f.EMPTY_METRIC_MEANS_UNKNOWN, }, metricCatalog: rows(f.METRIC_COUNT, (i) => ({ key: f.METRIC_KEYS[i], unit: f.METRIC_UNITS[i] })), diff --git a/apps/website/scripts/queen-wars-from-spec.test.mjs b/apps/website/scripts/queen-wars-from-spec.test.mjs index 4df558730e..3580916729 100644 --- a/apps/website/scripts/queen-wars-from-spec.test.mjs +++ b/apps/website/scripts/queen-wars-from-spec.test.mjs @@ -60,7 +60,7 @@ test('the WARS source compiles, evaluates its tests and renders all projections' assert.match(out.arena.configurations[1].source, /^https:\/\/docs\.typesafe\.ai\//) assert.equal(out.arena.experiments[0].issue.number, 4328) assert.equal(out.arena.experiments[0].modelEvidence, 'UNKNOWN') - assert.equal(out.arena.protocol.jevRoleEvidence, 'SOURCE-CLAIM') + assert.equal(out.arena.protocol.triRoleEvidence, 'OBSERVED') assert.equal(out.arena.runs.length, 1) assert.equal(out.arena.measurements.length, 5) }) diff --git a/apps/website/specs/queen/wars.t27 b/apps/website/specs/queen/wars.t27 index 503ea06e2a..a49983c9fb 100644 --- a/apps/website/specs/queen/wars.t27 +++ b/apps/website/specs/queen/wars.t27 @@ -25,16 +25,16 @@ pub const EMPTY_METRIC_MEANS_UNKNOWN : bool = true; // Fairness contract. A result that breaks one controlled factor is not an A/B result. pub const REAL_GITHUB_TASKS_ONLY : bool = true; -pub const VARIABLE_FACTOR : str = "jev-decision-layer"; +pub const VARIABLE_FACTOR : str = "tri-decision-layer"; pub const CONTROLLED_FACTORS : [7]str = ["issue-url", "base-sha", "executor-model", "prompt", "tool-policy", "budget", "acceptance"]; pub const ISOLATION : str = "one clean git worktree per arm, both pinned to the same base commit"; pub const ACCEPTANCE_POLICY : str = "the issue acceptance commands plus repository gates decide pass or fail"; pub const REVIEW_POLICY : str = "Queen reviews the diff and exact evidence after automated gates"; pub const WINNER_POLICY : str = "merge at most one accepted patch; preserve every arm in the append-only ledger"; pub const COMPARISON_VALIDITY_POLICY : str = "run both arms in one campaign with the same explicit executor model and reasoning configuration; an unpaired run or unknown model identity cannot determine a winner"; -pub const JEV_ROLE : str = "structured decision layer that ranks typed choices; never a chat model and never a code generator"; -pub const JEV_ROLE_EVIDENCE : str = "SOURCE-CLAIM"; -pub const JEV_ROLE_SOURCE : str = "https://docs.typesafe.ai/introduction/coding-agents"; +pub const TRI_ROLE : str = "the t27c compiler as judge: an arm is accepted only when the compiler and the acceptance gates accept its patch; it ranks typed choices and never generates the patch"; +pub const TRI_ROLE_EVIDENCE : str = "OBSERVED"; +pub const TRI_ROLE_SOURCE : str = "https://t27.ai/t27/files/trinity/apps/website/public/queen/wars.t27"; // Measurement vocabulary. Values are strings so an exact native unit is preserved; // absent rows mean unknown. A numeric zero is a measured zero only when a source exists. @@ -102,11 +102,11 @@ pub const MEASUREMENT_SOURCES : [5]str = ["session-observed isolated worktree co test protocol_has_one_variable { assert REAL_GITHUB_TASKS_ONLY == true; - assert VARIABLE_FACTOR == "jev-decision-layer"; + assert VARIABLE_FACTOR == "tri-decision-layer"; assert CONTROLLED_FACTORS[0] == "issue-url"; assert CONTROLLED_FACTORS[6] == "acceptance"; assert EMPTY_METRIC_MEANS_UNKNOWN == true; - assert JEV_ROLE_EVIDENCE == "SOURCE-CLAIM"; + assert TRI_ROLE_EVIDENCE == "OBSERVED"; } test arena_names_all_four_lanes { diff --git a/apps/website/src/components/QueenWars.tsx b/apps/website/src/components/QueenWars.tsx index aae24f8ef8..f783aa83a4 100644 --- a/apps/website/src/components/QueenWars.tsx +++ b/apps/website/src/components/QueenWars.tsx @@ -36,7 +36,7 @@ const COPY = { factor: 'variable', fixed: 'held constant', comparisonRule: 'comparison validity', - jevRole: 'JEV role', + triRole: 'TRI role (the judge)', task: 'REAL TASK', experimentPicker: 'EXPERIMENT', base: 'base', @@ -74,7 +74,7 @@ const COPY = { factor: 'переменная', fixed: 'зафиксировано', comparisonRule: 'валидность сравнения', - jevRole: 'роль JEV', + triRole: 'Роль TRI (судья)', task: 'РЕАЛЬНАЯ ЗАДАЧА', experimentPicker: 'ЭКСПЕРИМЕНТ', base: 'база', @@ -188,11 +188,11 @@ export function QueenWars({ lang }: { lang: 'en' | 'ru' }) {