From b946a2831243bd8481f7e5718ca7cd17118fce4a Mon Sep 17 00:00:00 2001 From: slaguru666 <111923774+slaguru666@users.noreply.github.com> Date: Sun, 13 Sep 2026 20:29:14 +0100 Subject: [PATCH] =?UTF-8?q?check-figures:=20the=20fold=20=E2=80=94=20one?= =?UTF-8?q?=20guard,=20three=20readers,=20and=20the=20boundary=20between?= =?UTF-8?q?=20them?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit check-unmarked is gone and check-figures holds all of it. Three readers, each authoritative where the corpus gives it authority: a named phrasing anywhere, a column header inside tables, and bold in prose only. The boundary is the finding rather than the union. Bold is a publication mark in prose and an emphasis mark in a table — nine measurement columns mix bold with plain, all correctly cited, and in the mixed-force table the two bolded rows are exactly the two the prose underneath singles out. So the bold reader stays silent in a table and the header rules there alone. Neither guard could have found this alone: each had half the evidence and read it as the other's bug. Union of both word lists, because each had a gap the other covered — wipes and runs. A proposal to drop runs? was made and withdrawn; it would have dropped the p99 column, which is R-299's own defect committed a second time. Exclusions test cells, never header words: prose, denominator, duration. No threshold rule touches a header, or "Past 15 rounds" loses two cited figures. Two integration defects, both caught by the ported tests: the readers stopped at different ends of one figure so the dedupe missed it (identity is where the number starts), and blanking prose cells for every reader dropped twelve real figures out of THROUGH_TRAIN's ratchet (the prose rule belongs to the column pass alone). Kept from R-299: opt-in rule, ratchet and its leave-the-loose-set branch, the UNCITEABLE check, NOT_A_MEASUREMENT, comments blanked not stripped. Added waivers, with an empty reason fatal and every waiver printed on a green build. 21 guards, 107 behavioural tests, 139 figures — 115 marked, 6 waived. Co-Authored-By: Claude Opus 5 --- docs/REVIEW_LOG.md | 76 +++++++++++++++++ package.json | 2 +- tools/check-behaviour.mjs | 90 +++++++++++++------- tools/check-figures.mjs | 172 ++++++++++++++++++++++++++++++++------ tools/check-unmarked.mjs | 150 --------------------------------- 5 files changed, 282 insertions(+), 208 deletions(-) delete mode 100644 tools/check-unmarked.mjs diff --git a/docs/REVIEW_LOG.md b/docs/REVIEW_LOG.md index bfd895f..f32b5eb 100644 --- a/docs/REVIEW_LOG.md +++ b/docs/REVIEW_LOG.md @@ -7652,3 +7652,79 @@ inspection, and seven columns asserted genuine inside the entry correcting exact **Nothing merged.** Both guards remain in the chain, both green, and the decision is still with the humans — but it is now a decision about whether to build, not about what to build. + +--- + +## R-307 — the fold: one guard, three readers, and the boundary between them + +**`check-unmarked` is gone and `check-figures` holds all of it.** Two guards covered one +question from R-299 to R-306; this is the merge, built in check-figures because the guard's +identity and its whole evidence chain already live there. + +**Three readers, each authoritative where the corpus gives it authority:** + +``` +shapes a named phrasing "a median 21 rounds" anywhere +columns the header's authority | Median rounds | 24 | tables only +bold the corpus's own mark **48.7%** prose only +``` + +**The boundary is the finding, not the union.** Bold is a publication mark in prose and an +emphasis mark in a table, and the corpus uses both correctly — nine measurement columns mix +bold with plain, and in the mixed-force table the two bolded rows are exactly the two the +prose underneath singles out. So the bold reader stays silent inside a table and the header +rules there alone. Neither of the two guards could have found this on its own: each of us had +one half of the evidence and read it as the other's bug. + +**The union of the word lists, because both halves were load-bearing.** `check-figures` could +not see *wipes* and missed the "Then wipes" column; `check-unmarked` could not see *runs* and +missed "1 in 100 runs past" — the p99 column R-269 added because the median is not a plan. A +proposal to drop `runs?` from the vocabulary was made here and withdrawn: it would have been +R-299's own defect committed a second time, by the session that found it. + +**Exclusions are tested on cells, never on header words.** *"1 in 100 runs past"* and +*"Measured over 300 runs"* both contain *runs*; one is a percentile in rounds and the other +is narrative. No edit to a word list separates them and looking at the cell does. Three now +exist, each from a case: a cell of **prose**, a **denominator** (`3.48 of 6`), and a +**duration** — the Act Two beat table has a column headed "Runs" holding `~10 min`, which is +how long a beat runs. The suite measures rounds and never minutes. + +⚠ **No threshold rule may touch a header.** It was proposed here and is wrong: *"Past 15 +rounds"* names a bucket and its cells are still measured percentages, so excluding the column +would drop two correctly-cited figures. + +### Two integration defects the merge created, both caught by the ported tests + +1. **A figure has two ends and the readers stopped at different ones.** `**8.3%**` is a bold + published figure and a `wipes N%` shape; one reader stopped after the digits and the other + after the asterisks, so the dedupe missed and it was reported twice. **The identity of a + figure is where its number starts.** +2. **An exclusion agreed and then applied in one of three places.** Blanking prose cells for + every reader read well and silently dropped twelve real figures out of THROUGH_TRAIN's + ratchet. The prose rule belongs to the **column** pass, whose claim rests on the header; + the shape pass matches a named phrasing and "2 of 4 hurt" is a measured figure wherever it + is written. Scoped back, and the ratchet count returned to 18. + +**Kept from R-299 unchanged**, all of it load-bearing: the opt-in rule, the ratchet and its +leave-the-loose-set branch, the UNCITEABLE check, `NOT_A_MEASUREMENT`, and comments blanked +rather than stripped — that last one is why the first draft of the header pass reported 130 +correctly-cited rows as unheld, because a citation marker is full of digits and none of them +are figures. Inline code is now blanked for the same reason. + +**Added: waivers.** A figure with no artifact carries ``. An empty reason +is fatal, the R-288 call, and every waiver prints on every green build so that a decision to +stop checking something stays visible rather than becoming the default. Six today, all named. + +**Proved by breaking it**, each case alone in a throwaway copy: + +| Break | Result | +|---|---| +| The historical bare `**24**` in the median-rounds table | fails at line 1277 | +| Strip markers from two **bold prose** figures — what R-299 alone missed | fails on both | +| Strip markers from five **unbolded cells** — what R-300 alone missed | fails on all five | +| Empty `uncited:` waiver | fails, naming the figure | +| Bare figure in a document that cites nothing | ratchet, not the build | + +**21 guards, 107 behavioural tests, 139 figures with 115 marked and 6 waived.** The reader +tests moved onto `scanDocument` and run on strings, so the shapes they name survive the file +that no longer exists. diff --git a/package.json b/package.json index 3c65c49..9e6de08 100644 --- a/package.json +++ b/package.json @@ -11,7 +11,7 @@ "play": "node tools/playthrough.mjs", "mj": "node tools/mj-queue.mjs", "simulate": "node tools/simulate.mjs", - "check": "bun tools/check-rules.mjs && bun tools/check-kits.mjs && bun tools/check-lang.mjs && bun tools/check-templates.mjs && bun tools/check-behaviour.mjs && bun tools/check-scenarios.mjs && bun tools/check-rollable.mjs && bun tools/check-outcomes.mjs && bun tools/check-creatures.mjs && bun tools/check-powers.mjs && bun tools/check-anatomy.mjs && bun tools/check-lethality.mjs && bun tools/check-focus.mjs && bun tools/check-firstblood.mjs && bun tools/check-attackers.mjs && bun tools/check-fight-tail.mjs && bun tools/check-packs.mjs && bun tools/check-cited.mjs && bun tools/check-unmarked.mjs && bun tools/check-figures.mjs && bun tools/check-handouts.mjs && bun tools/check-bestiary.mjs", + "check": "bun tools/check-rules.mjs && bun tools/check-kits.mjs && bun tools/check-lang.mjs && bun tools/check-templates.mjs && bun tools/check-behaviour.mjs && bun tools/check-scenarios.mjs && bun tools/check-rollable.mjs && bun tools/check-outcomes.mjs && bun tools/check-creatures.mjs && bun tools/check-powers.mjs && bun tools/check-anatomy.mjs && bun tools/check-lethality.mjs && bun tools/check-focus.mjs && bun tools/check-firstblood.mjs && bun tools/check-attackers.mjs && bun tools/check-fight-tail.mjs && bun tools/check-packs.mjs && bun tools/check-cited.mjs && bun tools/check-figures.mjs && bun tools/check-handouts.mjs && bun tools/check-bestiary.mjs", "test": "bun run check", "readme": "bun tools/update-readme.mjs" }, diff --git a/tools/check-behaviour.mjs b/tools/check-behaviour.mjs index db2ad5d..03a7fa9 100644 --- a/tools/check-behaviour.mjs +++ b/tools/check-behaviour.mjs @@ -32,7 +32,7 @@ import { ROLES, TRADES, INDUCTION, TIERS, TRADE_BANDS, CHARACTERISTIC_DICE } import { STARTER_AUTHORITY, STARTER_GROUPS, STARTER_CASE, STARTER_TEAM } from "./scenario-starter.mjs"; import { castMarkersIn, castLikeIn } from "./declared-cast.mjs"; -import { unmarkedIn } from "./check-unmarked.mjs"; +import { scanDocument } from "./check-figures.mjs"; import { tagsIn, classifiedFiles, playableFiles } from "./check-scenarios.mjs"; import { beatsIn, beatLikeIn } from "./outcome-coverage.mjs"; import { step5Split, split as step5Of, THING_RATING } from "./step5-split.mjs"; @@ -919,58 +919,88 @@ test("the blend uses raw proportions, not the rounded rows", () => { } }); -/* ----------------- check-unmarked, read on strings ----------------- */ +/* ------------- check-figures, the merged reader, read on strings ------------- */ /* On strings and not on the corpus, for the reason R-292 gives: a reader tested by editing the document proves the document, and goes quiet the day somebody rewords the sentence it - was anchored to. Every case below is a shape the reader must classify, stated outright. */ + was anchored to. Each case names the shape it classifies outright. */ -const um = md => unmarkedIn("x.md", md); +/* The strict regime is opt-in, so every fixture carries one real citation. */ +const OPTIN = "\n\nSee **13%** elsewhere.\n"; +const doc = md => scanDocument("x.md", md + OPTIN); +const firstFault = r => r.strict[0] ?? ""; -test("an unmarked bold percentage is fatal", () => - assert.equal(um("The party wipes **8.3%** of the time.").strict.length, 1)); +test("an unmarked bold percentage in PROSE is fatal", () => + assert.equal(doc("The party wipes **8.3%** of the time.").strict.length, 1)); -test("the same figure carrying a cite is not this guard's business", () => - assert.equal(um("wipes **8.3%** of the time").strict.length, 0)); +test("the same figure carrying a cite is held by check-cited, not reported here", () => + assert.equal(doc("wipes **8.3%** of the time").strict.length, 0)); + +test("a bold decimal in prose is a published figure too", () => + assert.equal(doc("Against what a table rolls the factor is **1.29**.").strict.length, 1)); test("a waiver clears the figure and goes on the record with its reason", () => { - const r = um("Dodge is **68%**."); + const r = doc("Dodge is **68%**."); assert.equal(r.strict.length, 0); assert.equal(r.waived.length, 1); assert.equal(r.waived[0].reason, "a rating, not a measurement"); }); -test("an EMPTY waiver is fatal — the R-288 call, again", () => - assert.equal(um("Dodge is **68%**.").strict.length, 1)); +test("an EMPTY waiver is fatal — the R-288 call, again", () => { + const r = doc("Dodge is **68%**."); + assert.equal(r.strict.length, 1); + assert.match(firstFault(r), /EMPTY/); +}); -test("a bold figure under a measurement header is fatal — the 24 itself", () => - assert.equal(um("| | Median rounds |\n|---|---|\n| six a side | **24** |").strict.length, 1)); +test("a bare cell under a measurement header is fatal — the 24 itself", () => + assert.equal(doc("| | Median rounds |\n|---|---|\n| six a side | 24 |").strict.length, 1)); -test("the same figure under a header that names no measurement is ignored", () => - assert.equal(um("| | Roll |\n|---|---|\n| it watches | **3** |").strict.length, 0)); +test("bold is NOT what makes a cell a figure — the same cell bolded still fails", () => + assert.equal(doc("| | Median rounds |\n|---|---|\n| six a side | **24** |").strict.length, 1)); + +test("a bold cell under a header naming no measurement is silent", () => + assert.equal(doc("| | Roll |\n|---|---|\n| it watches | **3** |").strict.length, 0)); + +test("a column whose cells are sentences is not a measurement column", () => + /* The COLUMN pass would otherwise claim the 3 on the header's authority alone. */ + assert.equal(doc("| | Measured over 300 runs |\n|---|---|\n| in cover | Untouched. " + + "A legitimate choice for a pass, seen 3 times. |").strict.length, 0)); + +test("but a named measurement phrasing in a prose cell is still a figure", () => + /* The SHAPE pass claims "2 of 4 hurt" wherever it is written. The prose rule is the + column pass's, because that is the reader whose claim rests on the header. */ + assert.ok(doc("| | Measured over 300 runs |\n|---|---|\n| in cover | 2 of 4 hurt, half a " + + "party member down. The intended shape. |").strict.length >= 1)); + +test("a duration column is not a harness figure — 'Runs' meaning how long a beat runs", () => + assert.equal(doc("| | Runs |\n|---|---|\n| the tests | ~15 min |").strict.length, 0)); + +test("a percentile column IS one — 'runs' must stay in the vocabulary", () => + assert.equal(doc("| | 1 in 100 runs past |\n|---|---|\n| the cut | 32.3 rounds |").strict.length, 1)); test("a fenced figure is the artifact, not a claim about it", () => - assert.equal(um("```\n wiped **8.3%**\n```").strict.length, 0)); + assert.equal(doc("```\n wiped **8.3%**\n```").strict.length, 0)); -test("an inline-code figure is ignored too", () => - assert.equal(um("see `wiped **8.3%**` above").strict.length, 0)); +test("an inline-code figure is quoted, not claimed", () => + assert.equal(doc("write it as `wiped **8.3%**` in the margin").strict.length, 0)); -test("a bare figure in a measured sentence is loose, never fatal", () => { - const r = um("The median across runs was 21 and it wiped 8.3% of parties."); - assert.equal(r.strict.length, 0); - assert.ok(r.loose.length >= 1, "a bare figure in a measured sentence must still be reported"); -}); +test("a bare prose figure matching a shape is caught by the shape pass", () => + assert.equal(doc("The fight ran a median 21 rounds.").strict.length >= 1, true)); -test("a skill rating in ordinary prose is neither strict nor loose", () => { - const r = um("Nia Mercer carries Pistol 55% and a bad temper."); - assert.equal(r.strict.length, 0); - assert.equal(r.loose.length, 0, "ratings are facts about a sheet, not measurements"); -}); +test("a skill rating in ordinary prose is not a figure", () => + assert.equal(doc("Nia Mercer carries Pistol 55% and a bad temper.").strict.length, 0)); test("a marker after the closing pipe of a table cell still counts", () => - assert.equal(um("| | Median rounds |\n|---|---|\n| six | **24** |").strict.length, 0)); + assert.equal(doc("| | Median rounds |\n|---|---|\n| six | 24 |").strict.length, 0)); + +test("a document that cites nothing is held by the ratchet, not the strict regime", () => { + const r = scanDocument("y.md", "The party wipes **8.3%** of the time."); + assert.equal(r.optedIn, false); + assert.equal(r.strict.length, 0, "an uncited document must not fail the build"); + assert.ok(r.unmarked >= 1, "but its bare figures are still counted for the ratchet"); +}); test("the reader reports the line the figure is actually on", () => - assert.equal(um("one\ntwo\nwipes **8.3%** here").strict[0].where, "x.md:3")); + assert.match(firstFault(doc("one\ntwo\nwipes **8.3%** here")), /x\.md:3/)); /* ---------------------------------------------------------------- */ diff --git a/tools/check-figures.mjs b/tools/check-figures.mjs index fe5e481..8ae9609 100644 --- a/tools/check-figures.mjs +++ b/tools/check-figures.mjs @@ -125,7 +125,21 @@ for (const rel of ["tools/pack-tables-baseline.json", "tools/fight-tail-baseline Drawn from the baselines' own field names, exactly as the shapes above are. A bare `%` is deliberately NOT here: a column of percentages may be skill ratings, and this corpus prints plenty of those. */ -const MEASURE_HEADER = /\b(median|mean|average|wiped?|wipes|hurt|down|deaths?|swing|rounds?)\b/i; +const MEASURE_HEADER = /\b(median|mean|average|wiped?|wipes|hurt|down|dead|deaths?|swing|rounds?|runs?|rate|odds|chance|share|percentile)\b/i; + +/* THE UNION OF TWO WORD LISTS, and both halves were load-bearing. This file could not see + *wipes* (`wiped?` does not match it) and missed the "Then wipes" column; check-unmarked + could not see *runs* and missed "1 in 100 runs past", which is the p99 column R-269 added + because the median is not a plan. Each list had a real gap the other covered, which is a + better argument for the union than any of the three rates we computed arguing about it. + + NO THRESHOLD EXCLUSION APPLIES TO A HEADER. It was proposed and it is wrong: "Past 15 + rounds" names a bucket and its cells are still measured percentages of runs, so excluding + the column would drop two correctly-cited figures. The threshold rule belongs where it + already is, on the cell. */ + +/** A cell holding prose rather than a value: a sentence break, or simply too many words. */ +const isProse = t => /[.!?]\s+\S/.test(t) || t.trim().split(/\s+/).filter(Boolean).length > 8; /** Split a markdown table row into cells, keeping each cell's offset in the line. */ function cells(line) { @@ -160,6 +174,12 @@ function tableHeaders(lines) { removed so every offset still points at the right character of the real line. */ const maskComments = line => line.replace(//g, m => " ".repeat(m.length)); +/* Inline code is quoted material, not the document making a claim — `wiped **8.3%**` in a + sentence about how a marker is written must not become a finding. Blanked, not stripped, + for the same reason the comments are: every offset has to keep pointing at the right + character of the real line, because the marker lookup runs against the unmasked text. */ +const maskCode = line => line.replace(/`[^`\n]*`/g, m => " ".repeat(m.length)); + /** Figures in a table row, by the header of their own column rather than by phrasing. */ export function cellFiguresIn(line, header) { if (!header) return []; @@ -168,17 +188,32 @@ export function cellFiguresIn(line, header) { const out = []; for (const [i, cell] of cells(masked).entries()) { if (!MEASURE_HEADER.test(heads[i]?.text ?? "")) continue; + /* A CELL THAT IS A SENTENCE IS NOT A VALUE. THROUGH_TRAIN's "Measured over 300 runs" + column holds prose — *"2 of 4 hurt, half a party member down, no wipes. The intended + shape."* — where the bold wraps a clause rather than a figure. The test is on the CELL + and not on the header's words, because "1 in 100 runs past" and "Measured over 300 + runs" both contain *runs* and only one of them is prose. No edit to a word list can + separate those two; looking at the cell can. */ + if (isProse(cell.text)) continue; const re = /(\d+(?:\.\d+)?)/g; let m; while ((m = re.exec(cell.text))) { /* "3.48 of 6" — the 6 is the party size the mean is out of, not a measurement. */ if (/\bof\s+$/i.test(cell.text.slice(0, m.index))) continue; + /* A DURATION IS NOT A HARNESS FIGURE. The Act Two beat table has a column headed + "Runs" holding `~10 min` — *how long the beat runs*, a different sense of the word + from the p99 column's "1 in 100 runs past". The suite measures ROUNDS and never + minutes; this document says so itself, calling three-to-four minutes a round "an + estimate, not a measurement". So the unit settles it, and the test is on the cell + because both headers contain *runs* and no word list can tell them apart. */ + if (/^\s*(?:min|mins|minutes?|hrs?|hours?|secs?|seconds?)\b/i.test(cell.text.slice(m.index + m[1].length))) continue; const numEnd = cell.start + m.index + m[1].length; out.push({ kind: `column:${(heads[i].text.replace(/[*|]/g, "").trim() || "?").toLowerCase()}`, value: m[1], text: cell.text.replace(/\s+/g, " ").trim() || m[1], at: numEnd, + numAt: numEnd - m[1].length, marked: MARKED_AFTER.test(line.slice(numEnd)) }); } @@ -186,12 +221,51 @@ export function cellFiguresIn(line, header) { return out; } +/* BOLD IS THE CORPUS'S OWN MARK OF A PUBLISHED FIGURE — IN PROSE, AND ONLY THERE. + The documents bold what they publish and leave ratings plain: `Brawl 55%` is a fact about + a sheet, `**8.3%**` is a claim about a run. That is a signal from the corpus's habits + rather than from a guard author's imagination, so it catches phrasings no shape list + anticipates — the two figures that survived every reader here were `**1.29**` and + `**48.7%**`, neither in a table and neither matching a shape. + + ⚠ AND IT IS WORTHLESS INSIDE A TABLE, which is the finding that sets this boundary. Nine + measurement columns in the corpus mix bold with plain, every cell correctly cited. Four + consecutive rows of the mixed-force table read `0%`, `**8.3%**`, `**30.1%**`, `93.8%` — + and the two bolded rows are exactly the two the prose beneath singles out. A table bolds + what it wants READ; prose bolds what it PUBLISHES. Same mark, two conventions, so inside a + table this pass stays silent and the column header rules alone. */ +const BOLD_PUBLISHED = /\*\*\s*(?:[~<>≈]\s*)?(\d[\d,]*(?:\.\d+)?\s*%|\d[\d,]*\.\d+)\s*\*\*/g; + +/** Bold figures in a line of prose. Returns nothing for a table row, on purpose. */ +export function boldFiguresIn(line) { + if (/^\s*\|/.test(line)) return []; + const masked = maskComments(line); + const out = []; + for (const m of masked.matchAll(BOLD_PUBLISHED)) { + const at = m.index + m[0].length; + out.push({ kind: "bold", value: m[1].replace(/[\s,%]/g, ""), text: m[0].trim(), at, + numAt: m.index + m[0].indexOf(m[1]), + marked: MARKED_AFTER.test(line.slice(at)) }); + } + return out; +} + /* The marker check, deliberately looser than check-cited's CITE. That reader is strict on purpose; here, ANY citation-shaped comment beside the figure means somebody marked it and check-cited owns the question of whether they marked it correctly. Two guards disagreeing about what counts as marked would leave a figure that neither of them holds. */ const MARKED_AFTER = /^[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*` instead. It must give + a reason: an empty waiver is a decision nobody recorded, and it is fatal — the call made + for the empty cast marker in R-288. Every waiver prints on every green build, so choosing + to stop checking something stays visible instead of becoming the default. */ +const WAIVED_AFTER = /^[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*/i; +const waiverAt = rest => { const m = WAIVED_AFTER.exec(rest); return m ? { reason: m[1] } : null; }; + /** Every measurement-shaped figure on one line, with whether it carries a marker. */ export function figuresIn(line) { const out = []; @@ -208,58 +282,92 @@ export function figuresIn(line) { the number, so it is counted once at the position it occupies. */ if (seen.has(numEnd)) continue; seen.add(numEnd); - out.push({ kind, value: m[1], text: m[0].trim(), marked: MARKED_AFTER.test(line.slice(numEnd)) }); + out.push({ kind, value: m[1], text: m[0].trim(), at: numEnd, numAt: numEnd - m[1].length, + marked: MARKED_AFTER.test(line.slice(numEnd)) }); } } return out.sort((a, b) => a.text.localeCompare(b.text)); } -const strict = []; -const loose = {}; -const rows = []; - -for (const [label, abs] of playableFiles()) { - const src = readFileSync(abs, "utf8"); +/** + * Every figure in one document, by all three readers. Exported and taking a STRING because + * a reader tested by editing the corpus proves the corpus, and goes quiet the day somebody + * rewords the sentence it was anchored to (R-292). + */ +export function scanDocument(label, src) { + const lines = src.split("\n"); /* Fenced blocks are the safe harbour, exactly as they are for check-outcomes: a code sample or a recorded harness dump is not the document making a claim. */ const fenced = new Set(); let inFence = false; - const lines = src.split("\n"); lines.forEach((ln, i) => { if (/^\s*```/.test(ln)) { inFence = !inFence; fenced.add(i); return; } if (inFence) fenced.add(i); }); const optedIn = /` in the corpus against the artifact it - * names, and it is exact. But it can only resolve the markers that EXIST. A number written - * without one is not checked, not reported, and not distinguishable from prose — and the - * guard's green line says "160 cited figures resolve" either way, which reads like coverage. - * - * That is how "a median 24 rounds" survived eleven desk passes in three places. Nothing was - * wrong with check-cited. The figure simply never entered its field of view. - * - * So this is the other half, and it is deliberately ASYMMETRIC — the shape this project - * arrived at for cast markers, beat tags and `POWER:`. The strict reader above says whether - * what is declared is right. This one says what was never declared at all: - * - * STRICT a published measurement — a bold percentage or decimal, or a bold number in a - * table whose own header calls the column a measurement — must carry `cite:`, - * or an `uncited:` waiver that says why. Anything else stops the build. - * - * LOOSE a bare percentage or decimal sitting in a sentence about medians, wipes or - * runs is COUNTED and named, never fatal. It is the class this guard knows it - * cannot judge: most are ratings and prose, some are the next stale 24. - * - * The waiver is not a silencer. It must carry a reason, an empty one is fatal (the call made - * for the empty cast marker in R-288), and every waiver is printed on every green build so - * that a decision to stop checking something stays visible rather than becoming the default. - * - * Skill ratings are not measurements. "Brawl 55%" is a fact about a sheet, not a figure - * derived from running the game, and bolding is how this corpus distinguishes them: the - * documents bold what they publish and leave ratings plain. That convention is the whole - * basis of the strict class, and it is why the loose class exists to catch its exceptions. - * - * node tools/check-unmarked.mjs check the corpus - * node tools/check-unmarked.mjs --loose also print every loose hit, not just the count - */ -import { readFileSync } from "node:fs"; -import { playableFiles } from "./check-scenarios.mjs"; - -const LOOSE_LIST = process.argv.slice(2).includes("--loose"); - -/* Code fences hold printed baselines and harness output — figures that are already the - artifact rather than a claim about it. Blanked rather than removed so every index still - points at the right line. */ -const CODE = /```[\s\S]*?```|`[^`\n]*`/g; -const scannable = text => text.replace(CODE, m => " ".repeat(m.length)); - -const PERCENT_OR_DECIMAL = String.raw`\d[\d,]*(?:\.\d+)?\s*%|\d[\d,]*\.\d+`; -const BOLD_MEASURED = new RegExp(String.raw`\*\*\s*(?:[~<>≈]\s*)?(${PERCENT_OR_DECIMAL})\s*\*\*`, "g"); -const BOLD_NUMBER = /\*\*\s*(?:[~<>≈]\s*)?(\d[\d,]*(?:\.\d+)?\s*%?)\s*\*\*/g; -const BARE_FIGURE = new RegExp(String.raw`(?]*?)\s*-->/; - -/* What makes a COLUMN a measurement. Note `%` is absent on purpose: a column of percentages - may be skill ratings, and the first draft of this guard called all 135 of them figures. */ -const MEASURE_WORD = /\b(median|mean|average|wiped?|deaths?|dead|rounds?|runs?|rate|odds|chance|share|percentile)\b/i; -/* The same test for a sentence, plus the phrasings prose uses that a column heading does not. */ -const MEASURED_SENTENCE = new RegExp(MEASURE_WORD.source + String.raw`|\bof the time\b|\bin \d+ (?:runs|sessions|seeds)\b`, "i"); - -/** Header of the markdown table a line belongs to, or null when it is not in one. */ -function tableHeaders(lines) { - const owner = new Array(lines.length).fill(null); - for (let i = 1; i < lines.length; i++) { - if (!/^\s*\|[\s:|-]+\|\s*$/.test(lines[i])) continue; // the |---|---| separator - if (!/^\s*\|/.test(lines[i - 1])) continue; - const header = lines[i - 1]; - for (let j = i + 1; j < lines.length && /^\s*\|/.test(lines[j]); j++) owner[j] = header; - } - return owner; -} - -const lineIndexOf = (text, at) => text.slice(0, at).split("\n").length - 1; - -export function unmarkedIn(label, raw) { - const text = scannable(raw); - const lines = raw.split("\n"); - const owner = tableHeaders(lines); - const strict = [], loose = [], waived = []; - const claimed = new Set(); - - const consider = (m, why) => { - const end = m.index + m[0].length; - const after = MARKER_AFTER.exec(raw.slice(end, end + 120)); - const li = lineIndexOf(raw, m.index); - const where = `${label}:${li + 1}`; - if (after) { - if (after[1] === "uncited") { - if (!after[2]) strict.push({ where, figure: m[1], why: - `carries an empty \`uncited:\` waiver. A waiver with no reason is a decision nobody ` - + `recorded — say why this figure is not citable, or cite it.` }); - else waived.push({ where, figure: m[1], reason: after[2] }); - } - claimed.add(`${li}:${m.index}`); - return; // a cite: is check-cited's business, not ours - } - strict.push({ where, figure: m[1], why, line: lines[li].trim() }); - }; - - for (const m of text.matchAll(BOLD_MEASURED)) - consider(m, "a published percentage or decimal carrying no marker"); - - for (const m of text.matchAll(BOLD_NUMBER)) { - const li = lineIndexOf(raw, m.index); - const head = owner[li]; - if (!head || !MEASURE_WORD.test(head)) continue; - if (PERCENT_OR_DECIMAL && new RegExp(`^(?:${PERCENT_OR_DECIMAL})$`).test(m[1].trim())) continue; - consider(m, `a bold figure in a table whose header reads "${head.replace(/\|/g, "").trim()}"`); - } - - for (const m of text.matchAll(BARE_FIGURE)) { - const li = lineIndexOf(raw, m.index); - if (claimed.has(`${li}:${m.index}`)) continue; - if (!MEASURED_SENTENCE.test(lines[li])) continue; - const end = m.index + m[0].length; - if (MARKER_AFTER.test(raw.slice(end, end + 120))) continue; - loose.push({ where: `${label}:${li + 1}`, figure: m[1], line: lines[li].trim() }); - } - return { strict, loose, waived }; -} - -if (import.meta.main || process.argv[1]?.endsWith("check-unmarked.mjs")) { - const strict = [], loose = [], waived = []; - for (const [label, abs] of playableFiles()) { - const r = unmarkedIn(label, readFileSync(abs, "utf8")); - strict.push(...r.strict); loose.push(...r.loose); waived.push(...r.waived); - } - - if (strict.length) { - console.error(`check-unmarked: FAILED — ${strict.length} published figure(s) carry no citation marker\n`); - for (const s of strict) { - console.error(` ${s.where} ${s.figure}`); - console.error(` ${s.why}`); - if (s.line) console.error(` ${s.line.slice(0, 96)}`); - } - console.error(`\n A figure without a marker is not checked by anything. Cite it against its`); - console.error(` artifact, or waive it with .`); - process.exit(1); - } - - if (waived.length) { - console.log(`check-unmarked: ${waived.length} figure(s) waived, each on the record:`); - for (const w of waived) console.log(` ${w.where} ${w.figure} — ${w.reason}`); - } - if (LOOSE_LIST) for (const l of loose) console.log(` loose ${l.where} ${l.figure} | ${l.line.slice(0, 80)}`); - - console.log(`check-unmarked: OK — every published figure in ${playableFiles().length} playable files ` - + `carries a marker, ${waived.length} waived, ${loose.length} unbolded figure(s) in measured ` - + `sentences left unjudged (--loose to list them)`); -}