diff --git a/docs/REVIEW_LOG.md b/docs/REVIEW_LOG.md index 38e0f30..2528662 100644 --- a/docs/REVIEW_LOG.md +++ b/docs/REVIEW_LOG.md @@ -6930,3 +6930,49 @@ hypothetical until a skill nobody on the roster reaches proved it. And the first new test ended in `assert.ok(… || true)` — a tautology that passes on anything, which is the decoration this suite exists to refuse. Mutation-checked after removing it: making `playableFiles` stop excluding records turns it red. + +## R-294 — a citation that resolves against the rule, not against a stored copy of it + +c0 asked whether the step-5 contest split should get a baseline, and argued against its own +suggestion: an artifact that is a pure function of `rules.mjs` is a cache of the rule, and a +guard over it mostly asserts that arithmetic has not changed. The objection is right. The +conclusion is not "do not guard it" but **"do not store it"**. + +`check-cited`'s `ARTIFACTS` map now takes a `{ derive }` entry as well as a file path. A +string is a recorded baseline — sampled, seeded, noisy, expensive to re-run, and re-recorded +with `--update`. A derived source is enumerated on this build. Nothing is stored, so nothing +goes stale and **there is no `--update` able to silence a real disagreement between the +document and the game.** The document writes `**48.1%**` and +does not care which kind it is. + +`tools/step5-split.mjs` is the first one: all 10,000 roll pairs against `opposedContestFor`, +no seed and no runs. **Targets derived, never written.** 55 and 60 are the lowest POW×5 in +the six and in the four-player cut; 42 is `applyDifficulty(85, "difficult")`. + +**Two rules produce those three numbers and c0 caught me before I collapsed them.** The +refused rows are "the lowest POW in the room" over a population. The accepted row is not the +lowest of anything — it is a named exception that takes Ashcroft and drops him to Difficult. +One derivation covering both would be right today for the wrong reason, and there is now a +test that fails if someone unifies them. + +**A tie is fatal**, also c0's: "the lowest POW in the room" stops being well-defined the +moment two agents share it, and answering anyway would hold the prose against an arbitrary +pick. This fired for real during testing — raising Okonkwo's POW by one ties him with +Braithwaite at 60, and the build stopped with both names. + +**Proved four ways** in worktrees: a correct citation resolves (29 figures, 5 sources); a +wrong one fails with its own message, which says *nothing was re-recorded and nothing can +be*; a tie refuses; and raising Okonkwo's POW to 13 moves the lowest to Braithwaite, changes +`refused.held` from 48.1 to 52.4, and fails the citation that was correct a moment earlier. +That last one is the whole point — the figure follows the rule. + +**Two errors of mine, both caught by measurement rather than review.** The blend was computed +from the rounded rows under a comment saying it used raw proportions, giving 43.7 where the +document correctly prints 43.6; a comment describing what code ought to do is the defect this +suite is named for, and this one lasted ninety seconds. And the first exhaustiveness test +asserted the displayed percentages sum to 100 — they need not: 40.7 + 48.1 + 11.3 is 100.1, +three independent round-ups, and the document's table carries the same harmless 0.1. The code +was right and the test was wrong. + +**Landed unused, deliberately.** No citation markers were placed, because `CLEAN_GROUND.md` is +c0's and the markers are theirs to place. The mechanism is what this commit is. diff --git a/tools/check-behaviour.mjs b/tools/check-behaviour.mjs index c3020b8..cbcca00 100644 --- a/tools/check-behaviour.mjs +++ b/tools/check-behaviour.mjs @@ -34,6 +34,7 @@ import { STARTER_AUTHORITY, STARTER_GROUPS, STARTER_CASE, STARTER_TEAM } import { castMarkersIn, castLikeIn } from "./declared-cast.mjs"; import { tagsIn, classifiedFiles, playableFiles } from "./check-scenarios.mjs"; import { beatsIn, beatLikeIn } from "./outcome-coverage.mjs"; +import { step5Split, split as step5Of, THING_RATING } from "./step5-split.mjs"; let passed = 0, failed = 0; // Async tests are awaited in order rather than fired off — a failure that lands after @@ -870,6 +871,53 @@ test("the corpus knows which of its documents are played and which are records", assert.ok(!playable.some(l => /_ART\.md$/.test(l)), "an art prompt sheet has no rolls"); }); +/* ── R-294: the step-5 split, derived rather than stored ─────────────────────────────── */ + +test("the step-5 outcomes are exhaustive — every contest lands somewhere", () => { + /* Asserted on the RAW proportions, because that is where the invariant lives. The first + version of this test checked the rounded percentages and failed at target 55: 40.7 + + 48.1 + 11.3 is 100.1, three independent round-ups. The code was right and the test was + wrong, which is worth a comment — the displayed parts of a split need not sum to 100, + and the document's own table has the same harmless 0.1. */ + for (const target of [42, 55, 60, 75]) { + const s = step5Of(THING_RATING, target); + const raw = s.raw.taken + s.raw.held + s.raw.unsettled; + assert.equal(Math.round(raw * 10000), 10000, + `every one of the 10,000 pairs must land in exactly one outcome, at target ${target}`); + const shown = s.taken + s.held + s.unsettled; + assert.ok(Math.abs(shown - 100) <= 0.2, + `rounded parts drifted ${shown} from 100 at target ${target} — more than independent ` + + `rounding explains`); + } +}); + +test("the accepted row is a named exception, not the lowest POW in the room", () => { + /* Two different rules produce these numbers. Collapsing them into one derivation would + give an answer that is right today for the wrong reason — c0's objection, kept as a + test so the next person to tidy this cannot quietly unify them. */ + const s = step5Split(); + assert.ok(s.accepted.target < s.refused.target, + "Ashcroft at Difficult must resist lower than the lowest POW in the room, or the " + + "accepted branch is not the exception the scenario describes"); + assert.notEqual(s.accepted.who, s.refused.who); +}); + +test("the blend uses raw proportions, not the rounded rows", () => { + /* THE BUG THIS FILE WATCHES FOR, made in this very module and caught by comparing against + the document: blending 40.7 and 48.7 gives 43.7 where the answer is 43.637 -> 43.6. The + comment above the code said "raw proportions" while the code used the rounded ones. */ + const s = step5Split(); + const w = s.refusedShare / 100; + const fromRaw = Math.round((w * s.refused.raw.taken + (1 - w) * s.accepted.raw.taken) * 1000) / 10; + assert.equal(s.blended.taken, fromRaw, "blended.taken must be computed from the raw proportions"); + const fromRounded = Math.round((w * s.refused.taken + (1 - w) * s.accepted.taken) * 10) / 10; + if (fromRounded !== fromRaw) { + assert.notEqual(s.blended.taken, fromRounded, + "this is the rounded-input blend, which is the defect; if the two ever coincide this " + + "assertion goes quiet, and the equality above is the one that still holds"); + } +}); + /* ---------------------------------------------------------------- */ await runAll(); diff --git a/tools/check-cited.mjs b/tools/check-cited.mjs index d12d58e..f80e7a4 100644 --- a/tools/check-cited.mjs +++ b/tools/check-cited.mjs @@ -24,19 +24,39 @@ import { readFileSync, existsSync } from "node:fs"; import path from "node:path"; import { scenarioFiles } from "./check-scenarios.mjs"; +import { step5Split } from "./step5-split.mjs"; const ROOT = path.resolve(path.dirname(new URL(import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1")), ".."); const LIST = process.argv.slice(2).includes("--list"); -/* Artifact name as written in a citation -> the file it means. Adding a baseline here is - what makes it citeable; nothing else in the document needs to know. */ +/* Artifact name as written in a citation -> where the figure comes from. Adding an entry + here is what makes a figure citeable; nothing else in the document needs to know. + + TWO KINDS OF SOURCE, because there are two kinds of figure and they fail differently. + + A STRING is a recorded baseline: a sampled measurement with a seed, a run count and noise + around it. Re-running is expensive, so the number is stored and `--update` re-records it. + The failure such a citation catches is "somebody re-recorded and left the prose behind". + + A `{ derive }` is computed on this build. It is for a figure that is a pure function of + rules.mjs — enumerated, exact, no seed and no runs. Storing one would be a cache of the + rule, and a cache can go stale; worse, it would arrive with an `--update` able to silence + a real disagreement between the document and the game. So nothing is stored, there is + nothing to update, and the failure it catches is the one that actually happened three + times to the same number: the prose is wrong about the rule. + + The mechanism is identical either way, which is the point. The document writes + `**48.1%**` and does not care which kind it is. */ const ARTIFACTS = { "first-blood": "tools/first-blood-baseline.json", "attackers": "tools/attackers-baseline.json", "fight-tail": "tools/fight-tail-baseline.json", "lethality": "tools/lethality-baseline.json", - "outcomes": "tools/outcome-coverage-baseline.json" + "outcomes": "tools/outcome-coverage-baseline.json", + "step5": { derive: step5Split, label: "tools/step5-split.mjs, enumerated on this build" } }; +const isDerived = name => typeof ARTIFACTS[name] === "object"; +const sourceOf = name => isDerived(name) ? ARTIFACTS[name].label : ARTIFACTS[name]; /* Fields that exist in an artifact but must never appear in prose. A sample maximum is the clearest case: same party, same seeds, same runs, and renaming a config moved @@ -51,9 +71,16 @@ const UNCITEABLE = { const loaded = new Map(); function artifact(name) { if (loaded.has(name)) return loaded.get(name); - const rel = ARTIFACTS[name]; - if (!rel) return null; - const file = path.join(ROOT, rel); + const entry = ARTIFACTS[name]; + if (!entry) return null; + /* Derived once per run, not once per citation: a dozen citations against one enumeration + should cost one enumeration. */ + if (isDerived(name)) { + const value = entry.derive(); + loaded.set(name, value); + return value; + } + const file = path.join(ROOT, entry); const json = existsSync(file) ? JSON.parse(readFileSync(file, "utf8")) : null; loaded.set(name, json); return json; @@ -152,8 +179,13 @@ for (const [rel, abs] of scenarioFiles()) { } rows.push([rel, `${name} ${dotted}`, printed, String(held)]); if (Number(printed) !== Number(held)) { - problems.push(` ${rel}: prints ${printed} but ${ARTIFACTS[name]} holds ${held} ` - + `(${dotted}). Re-recording updated the artifact and left the prose behind — fix the sentence.`); + problems.push(isDerived(name) + ? ` ${rel}: prints ${printed} but the game gives ${held} (${dotted}, from ` + + `${sourceOf(name)}). Nothing was re-recorded and nothing can be — this figure is ` + + `enumerated from rules.mjs on every build, so the document is wrong about the ` + + `rule. Fix the sentence, or change the rule and watch the number follow.` + : ` ${rel}: prints ${printed} but ${sourceOf(name)} holds ${held} ` + + `(${dotted}). Re-recording updated the artifact and left the prose behind — fix the sentence.`); } } } diff --git a/tools/step5-split.mjs b/tools/step5-split.mjs new file mode 100644 index 0000000..b480fe4 --- /dev/null +++ b/tools/step5-split.mjs @@ -0,0 +1,150 @@ +/** + * The Act Four countdown's step 5, enumerated rather than sampled. + * + * `opposedContestFor` returns THREE outcomes and CLEAN GROUND described two for a long + * while: the thing takes the person, the person holds, or the contest settles nothing. + * The held case is the likeliest single result at most tables and went unwritten through + * several versions. Worse, the figures that WERE written drifted — a 74.3% quoted as + * something a GM meets when it is a contest that never happens, and a "nearly quadruples" + * built on the same counterfactual. Both were found by re-deriving them by hand. + * + * THIS FILE EXISTS SO NOBODY HAS TO DO THAT BY HAND AGAIN, and so check-cited can hold the + * prose to it. It is a pure function of rules.mjs and the roster: no seed, no runs, no + * noise. That is why there is no baseline JSON beside it and no `--update` — a stored copy + * would be a cache of the rule, and a cache is a thing that can go stale. The figures are + * recomputed on every build and the document is checked against what the game actually + * does today. + * + * node tools/step5-split.mjs print the split, with its derivation + */ +import { opposedContestFor, applyDifficulty } from "../rules.mjs"; +import { ROSTER } from "./roster.mjs"; +import { expandFromRegister } from "./expand-spec.mjs"; +import { castAndCut } from "./declared-cast.mjs"; + +/* The thing reaches at 75, not 100. This is the one number here with no source in the + engine — it is a scenario constant — so it is named once, and check-cited asserts the + document still says 75 rather than letting the two drift apart silently. */ +export const THING_RATING = 75; + +/** POW×5, the resistance an agent brings to step 5. */ +const powTimesFive = key => { + const spec = ROSTER.find(r => r.key === `pc_${key}`) ?? ROSTER.find(r => r.key === key); + if (!spec) throw new Error(`step5-split: "${key}" is not on the duty roster`); + const e = expandFromRegister(spec); + return { who: e.name, rating: (e.ch?.pow ?? 0) * 5 }; +}; + +/* A TIE IS FATAL, NOT A COIN TOSS. "The lowest POW in the room" stops being well-defined + the moment two agents share it, and a resolver that quietly takes the first would hold the + prose against an arbitrary pick while the scenario's own instruction had become ambiguous. + The failure there is not the wrong answer, it is answering at all — the same call as the + empty cast marker in R-288. Refuse, name both, and let whoever re-cast the roster decide + what the scenario should say. (c0 raised this; today okonkwo is uniquely 55 and + braithwaite uniquely 60, so it costs nothing until it matters.) */ +const lowest = keys => { + const rated = keys.map(powTimesFive).sort((a, b) => a.rating - b.rating); + const tied = rated.filter(r => r.rating === rated[0].rating); + if (tied.length > 1) { + console.error(`step5-split: ${tied.map(t => t.who).join(" and ")} both have POW×5 ` + + `${rated[0].rating}, so "the lowest POW in the room" names two people. The countdown ` + + `cannot be enumerated until the scenario says which, and no figure derived from a ` + + `silent choice between them would be worth printing.`); + process.exit(1); + } + return rated[0]; +}; + +/** + * All 10,000 (active, resisting) roll pairs. Enumerated, not sampled: every pair is + * weighted equally and occurs exactly once, so these are the odds and not an estimate + * of them. Percentages to one decimal, which is how the document writes them. + */ +export function split(activeRating, resistingRating) { + let taken = 0, held = 0, unsettled = 0; + for (let a = 1; a <= 100; a++) { + for (let r = 1; r <= 100; r++) { + const o = opposedContestFor({ rating: activeRating, roll: a }, + { rating: resistingRating, roll: r }); + if (!o.settled) unsettled++; + else if (o.winner === "active") taken++; + else held++; + } + } + const pc = n => Math.round(n / 10000 * 1000) / 10; + /* `raw` carries the unrounded proportions because the blend below needs them. Returning + only the rounded figures is what made the first version of this file blend 40.7 and 48.7 + into 43.7 when the answer is 43.637 — under a comment claiming it used raw proportions. + A comment that describes what the code ought to do is the defect this project is named + for, and this one lasted about ninety seconds. */ + return { taken: pc(taken), held: pc(held), unsettled: pc(unsettled), + raw: { taken: taken / 10000, held: held / 10000, unsettled: unsettled / 10000 } }; +} + +/** + * The cases that actually occur, which is the distinction the document kept losing. + * + * Act Two decides which branch a table is in. If Ashcroft REFUSES, the countdown never + * reaches for him — it reaches for the lowest POW in the room. If he ACCEPTS, it takes him + * and he resists at Difficult. There is no case in which the thing reaches for a refusing + * Ashcroft, so there is no figure for one. + * + * TWO DIFFERENT RULES PRODUCE THESE THREE NUMBERS, and collapsing them into one derivation + * would give an answer that is right today for the wrong reason. The refused rows are + * "the lowest POW in the room" over a population. The accepted row is not the lowest of + * anything — it is a named exception that takes Ashcroft specifically and drops him to + * Difficult. So `lowest()` is used for the first two and never for the third. + * + * The targets are derived, never written down: change a POW on the roster and these move, + * and the prose has to follow. A literal 55 here would agree with a stale document forever. + * + * There are three table sizes and only two rows, which is not an omission. The scaling note + * drops Pollard at five and Okonkwo as well at four; at five the lowest POW is still + * Okonkwo 55, so the five-player figures are identical to the six-player ones and the + * document is right to print no separate row for them. + */ +export function step5Split() { + const { line, cut } = castAndCut("step5-split"); + const six = lowest(line); + const four = lowest(cut); + const ash = powTimesFive("ashcroft"); + const accepted = { who: ash.who, rating: applyDifficulty(ash.rating, "difficult") }; + + /* He refuses on a plain Insight success, so the branch weights are that rating. */ + const spec = ROSTER.find(r => r.key === "pc_ashcroft"); + const insight = (expandFromRegister(spec).skills ?? []) + .find(s => s.fam === "insight")?.val ?? 0; + + const rows = { + refused: { ...six, target: six.rating, who: six.who }, + cut: { ...four, target: four.rating, who: four.who }, + accepted: { ...accepted, target: accepted.rating, who: accepted.who } + }; + for (const k of ["refused", "cut", "accepted"]) { + Object.assign(rows[k], split(THING_RATING, rows[k].target)); + } + + /* Blended from the RAW proportions, not from the rounded rows — weighting rounded + percentages is how 43.637 becomes 43.7, and it cost a wrong correction once. */ + const w = insight / 100; + const mix = k => Math.round((w * rows.refused.raw[k] + (1 - w) * rows.accepted.raw[k]) * 1000) / 10; + const blended = { taken: mix("taken"), held: mix("held"), unsettled: mix("unsettled") }; + + return { ...rows, blended, refusedShare: insight, acceptedShare: 100 - insight, + thing: THING_RATING }; +} + +if (import.meta.main || process.argv[1]?.endsWith("step5-split.mjs")) { + const s = step5Split(); + console.log(`\n The thing reaches at ${s.thing}. Ashcroft refuses on Insight, so ` + + `${s.refusedShare}% of tables are in the first row and ${s.acceptedShare}% the second.\n`); + console.log(" case resists taken held unsettled"); + for (const k of ["refused", "cut", "accepted"]) { + const r = s[k]; + console.log(` ${k.padEnd(9)} ${(r.who + " " + r.target).padEnd(18)} ` + + `${String(r.taken).padStart(5)} ${String(r.held).padStart(6)} ${String(r.unsettled).padStart(9)}`); + } + const b = s.blended; + console.log(` ${"blended".padEnd(9)} ${"across all games".padEnd(18)} ` + + `${String(b.taken).padStart(5)} ${String(b.held).padStart(6)} ${String(b.unsettled).padStart(9)}\n`); +}