/** * A figure nothing holds is a figure that drifts, and the guard that would catch it cannot * see it. * * `check-cited` resolves every citation marker in the corpus against the artifact it names. * It is a good reader and it has exactly one blind spot, which is structural rather than a * bug: it can only resolve the markers that exist. A measured figure written into prose with * NO marker beside it is not a failed citation — it is not a citation at all, and nothing * looks at it ever again. * * That is not hypothetical. Desk pass 11 found "a median 24 rounds" in the Pacing Note and * "24 if the column joins" in EXPOSURE. The baseline holds 21 at two of the column joining * and 25 at three. 24 is neither, it had been wrong through every pass that checked * citations, and it survived because it carried no marker to check. The same document * printed its four-player wipe rate uncited twice while citing it correctly four times — * and that figure is the one that was published at 58.7% until a truncated sweep was found. * * ⚠ THE SHAPES HAVE A BLIND SPOT AND IT IS STRUCTURAL. R-299 said this guard "learns its * vocabulary from the marked figures and applies it to the unmarked" — and a vocabulary * learned that way cannot contain the phrasing of the figure nobody marked. The concurrent * session proved it against a live defect: a THIRD stale 24, in a table cell, reported OK by * this file. Its measurement status lived in the header row two lines above — * * | | Median rounds | * |---|---| * | Six hollow men + three of the column | **24** | <- the baseline says 25 * * — and the scan read one line at a time, so a bare number in a column the document itself * calls "Median rounds" was invisible. Fixed below by reading the header: a cell inherits * the measurement status of its column, which is a claim the DOCUMENT makes rather than one * this file's vocabulary has to anticipate. That is the narrow repair. The general point * stands and is worth keeping in view — see `check-unmarked`, which attacks the same problem * from the other end by treating bold as the corpus's own mark of a published figure. * * So this is the asymmetric counter for the other direction, and the fifth in this repo: * * check-cited reads markers -> is the figure beside this marker right? * check-figures reads FIGURES -> does this figure have a marker at all? * * THE OPT-IN RULE. A document that uses citations has opted into the regime, and every * measurement-shaped figure in it must carry a marker. A document with no markers anywhere * has not, and its figures are counted and held by a ratchet rather than failing the build — * because turning three unguarded scenarios red is how a guard gets switched off on the day * it is written. The ratchet still means they cannot get worse, and the moment such a * document gains its first citation it joins the strict regime in full. * * node tools/check-figures.mjs check the corpus * node tools/check-figures.mjs --list print every figure found, marked or not * node tools/check-figures.mjs --update re-record the ratchet for opted-out documents */ import { readFileSync, writeFileSync, existsSync } from "node:fs"; import path from "node:path"; import { playableFiles } from "./check-scenarios.mjs"; const ROOT = path.resolve(path.dirname(new URL(import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1")), ".."); const BASELINE = path.join(ROOT, "tools", "figures-baseline.json"); const argv = process.argv.slice(2); const LIST = argv.includes("--list"); const UPDATE = argv.includes("--update"); /* A number, with the document's own bolding allowed around it. */ const N = String.raw`\*{0,2}(\d+(?:\.\d+)?)\*{0,2}`; /* THE SHAPES A HARNESS FIGURE TAKES IN THIS CORPUS. Deliberately narrow. An earlier draft matched any number within forty-five characters of a measurement word and reported 282 candidates, nearly all of them prose — "down 140 steps", "an engineer on his rounds", "01:06". A guard that cries wolf 282 times is a guard nobody runs. These are the phrasings the documents actually use when quoting the simulator, and each one was read off a figure that IS cited somewhere. */ /* `d` (hasIndices) so each match reports where its capture actually sits — see numEnd. */ const SHAPES = [ ["median", new RegExp(String.raw`median(?:\s+of)?\s+` + N, "gid")], ["rounds", new RegExp(N + String.raw`\s*\*{0,2}\s*rounds\b`, "gid")], ["wiped", new RegExp(N + String.raw`\s*%?\s*\*{0,2}\s*(?:wiped|wipes)\b`, "gid")], ["wipes-pct", new RegExp(String.raw`wipe[sd]?\s+` + N + String.raw`\s*%`, "gid")], ["wipe-rate", new RegExp(String.raw`wipe rate[^.]{0,24}?` + N + String.raw`\s*%`, "gid")], ["of-N-hurt", new RegExp(N + String.raw`\s*\*{0,2}\s*of\s+\d+\s+(?:hurt|down)\b`, "gid")], ["deaths", new RegExp(N + String.raw`\s*\*{0,2}\s*deaths?\b`, "gid")], ["pct-of-runs", new RegExp(N + String.raw`\s*%[^.]{0,18}\bruns?\b`, "gid")] ]; /* WHAT IS NOT A MEASUREMENT, each with the case that put it here. A DENOMINATOR. "one party in 300 wiped" is a rate written longhand; 300 is the bottom of a fraction, not a sample statistic, and THROUGH_TRAIN writes rates that way throughout. A THRESHOLD. "Past 15 rounds" is a column header in fight-tail's table — it names the bucket the figures below are counted into, and it comes from the tool's own definition rather than from a run. Same for "1 in 100 runs past". BOTH ARE TESTED AGAINST THE CONTEXT, NOT THE MATCH. The first draft tested the matched text alone, which cannot work: the `rounds` shape captures "15 rounds" and the word that excuses it — "Past" — is outside the match. It reported a column header as an unheld figure, which is the false positive that makes a guard get switched off. */ const NOT_A_MEASUREMENT = [ [/\b(?:in|of)\s+\*{0,2}\d+(?:\.\d+)?\*{0,2}\s*%?\s*\*{0,2}\s*(?:wiped|wipes)\b/i, "a denominator — 'one party in N wiped' is a rate written longhand"], [/\b(?:past|beyond|over|under|at least|fewer than)\s+\*{0,2}\d+(?:\.\d+)?\*{0,2}\s*(?:%|\s*rounds?)?\s*$/i, "a threshold, not a sample — it names a bucket, not a result"] ]; /* FIELDS check-cited REFUSES TO LET ANYBODY CITE, and their values. `longest` is a sample maximum: same party, same seeds, same runs, and renaming a config moved one from 71 to 90 rounds, because the stream is derived from the id. check-cited refuses a citation against it for that reason — but refusing the CITATION while the document prints the NUMBER leaves the least stable figure in the suite as the only one nothing holds. So a bare figure whose value matches an uncitable field is reported as its own kind of problem, with the field named. */ const UNCITEABLE_FIELDS = ["longest"]; const uncitableValues = new Map(); for (const rel of ["tools/pack-tables-baseline.json", "tools/fight-tail-baseline.json"]) { const abs = path.join(ROOT, rel); if (!existsSync(abs)) continue; const walk = (o, trail) => { if (o && typeof o === "object") { for (const [k, v] of Object.entries(o)) { if (UNCITEABLE_FIELDS.includes(k) && typeof v === "number") { uncitableValues.set(String(v), `${rel} ${[...trail, k].join(".")}`); } walk(v, [...trail, k]); } } }; walk(JSON.parse(readFileSync(abs, "utf8")), []); } /* WHAT MAKES A COLUMN A MEASUREMENT, and therefore every number in it a figure. Drawn from the baselines' own field names, exactly as the shapes above are. A bare `%` is deliberately NOT here: a column of percentages may be skill ratings, and this corpus prints plenty of those. */ const MEASURE_HEADER = /\b(median|mean|average|wiped?|wipes|hurt|down|dead|deaths?|swing|rounds?|runs?|rate|odds|chance|share|percentile)\b/i; /* THE UNION OF TWO WORD LISTS, and both halves were load-bearing. This file could not see *wipes* (`wiped?` does not match it) and missed the "Then wipes" column; check-unmarked could not see *runs* and missed "1 in 100 runs past", which is the p99 column R-269 added because the median is not a plan. Each list had a real gap the other covered, which is a better argument for the union than any of the three rates we computed arguing about it. NO THRESHOLD EXCLUSION APPLIES TO A HEADER. It was proposed and it is wrong: "Past 15 rounds" names a bucket and its cells are still measured percentages of runs, so excluding the column would drop two correctly-cited figures. The threshold rule belongs where it already is, on the cell. */ /** A cell holding prose rather than a value: a sentence break, or simply too many words. */ const isProse = t => /[.!?]\s+\S/.test(t) || t.trim().split(/\s+/).filter(Boolean).length > 8; /** Split a markdown table row into cells, keeping each cell's offset in the line. */ function cells(line) { const out = []; let at = line.indexOf("|"); if (at < 0) return out; at += 1; for (;;) { const end = line.indexOf("|", at); if (end < 0) break; out.push({ text: line.slice(at, end), start: at }); at = end + 1; } return out; } /** For every line, the header row of the table it is a body row of — or null. */ function tableHeaders(lines) { const owner = new Array(lines.length).fill(null); for (let i = 1; i < lines.length; i++) { if (!/^\s*\|[\s:|-]+\|\s*$/.test(lines[i])) continue; /* the |---|---| separator */ if (!/^\s*\|/.test(lines[i - 1])) continue; for (let j = i + 1; j < lines.length && /^\s*\|/.test(lines[j]); j++) owner[j] = lines[i - 1]; } return owner; } /* A CITATION MARKER IS FULL OF DIGITS AND NONE OF THEM ARE FIGURES. `packs line.hollow6.hurt` holds a 6; `fight-tail cut.over15` holds a 15. Scanning raw cell text made the guard report every correctly-cited row in the document as unheld — 130 false positives, on the tables that are the best-marked thing in the corpus. Comments are BLANKED rather than removed so every offset still points at the right character of the real line. */ const maskComments = line => line.replace(//g, m => " ".repeat(m.length)); /* Inline code is quoted material, not the document making a claim — `wiped **8.3%**` in a sentence about how a marker is written must not become a finding. Blanked, not stripped, for the same reason the comments are: every offset has to keep pointing at the right character of the real line, because the marker lookup runs against the unmasked text. */ const maskCode = line => line.replace(/`[^`\n]*`/g, m => " ".repeat(m.length)); /** Figures in a table row, by the header of their own column rather than by phrasing. */ export function cellFiguresIn(line, header) { if (!header) return []; const heads = cells(header); const masked = maskComments(line); const out = []; for (const [i, cell] of cells(masked).entries()) { if (!MEASURE_HEADER.test(heads[i]?.text ?? "")) continue; /* A CELL THAT IS A SENTENCE IS NOT A VALUE. THROUGH_TRAIN's "Measured over 300 runs" column holds prose — *"2 of 4 hurt, half a party member down, no wipes. The intended shape."* — where the bold wraps a clause rather than a figure. The test is on the CELL and not on the header's words, because "1 in 100 runs past" and "Measured over 300 runs" both contain *runs* and only one of them is prose. No edit to a word list can separate those two; looking at the cell can. */ if (isProse(cell.text)) continue; const re = /(\d+(?:\.\d+)?)/g; let m; while ((m = re.exec(cell.text))) { /* "3.48 of 6" — the 6 is the party size the mean is out of, not a measurement. */ if (/\bof\s+$/i.test(cell.text.slice(0, m.index))) continue; /* A DURATION IS NOT A HARNESS FIGURE. The Act Two beat table has a column headed "Runs" holding `~10 min` — *how long the beat runs*, a different sense of the word from the p99 column's "1 in 100 runs past". The suite measures ROUNDS and never minutes; this document says so itself, calling three-to-four minutes a round "an estimate, not a measurement". So the unit settles it, and the test is on the cell because both headers contain *runs* and no word list can tell them apart. */ if (/^\s*(?:min|mins|minutes?|hrs?|hours?|secs?|seconds?)\b/i.test(cell.text.slice(m.index + m[1].length))) continue; const numEnd = cell.start + m.index + m[1].length; out.push({ kind: `column:${(heads[i].text.replace(/[*|]/g, "").trim() || "?").toLowerCase()}`, value: m[1], text: cell.text.replace(/\s+/g, " ").trim() || m[1], at: numEnd, numAt: numEnd - m[1].length, marked: MARKED_AFTER.test(line.slice(numEnd)) }); } } return out; } /* BOLD IS THE CORPUS'S OWN MARK OF A PUBLISHED FIGURE — IN PROSE, AND ONLY THERE. The documents bold what they publish and leave ratings plain: `Brawl 55%` is a fact about a sheet, `**8.3%**` is a claim about a run. That is a signal from the corpus's habits rather than from a guard author's imagination, so it catches phrasings no shape list anticipates — the two figures that survived every reader here were `**1.29**` and `**48.7%**`, neither in a table and neither matching a shape. ⚠ AND IT IS WORTHLESS INSIDE A TABLE, which is the finding that sets this boundary. Nine measurement columns in the corpus mix bold with plain, every cell correctly cited. Four consecutive rows of the mixed-force table read `0%`, `**8.3%**`, `**30.1%**`, `93.8%` — and the two bolded rows are exactly the two the prose beneath singles out. A table bolds what it wants READ; prose bolds what it PUBLISHES. Same mark, two conventions, so inside a table this pass stays silent and the column header rules alone. */ const BOLD_PUBLISHED = /\*\*\s*(?:[~<>≈]\s*)?(\d[\d,]*(?:\.\d+)?\s*%|\d[\d,]*\.\d+)\s*\*\*/g; /** Bold figures in a line of prose. Returns nothing for a table row, on purpose. */ export function boldFiguresIn(line) { if (/^\s*\|/.test(line)) return []; const masked = maskComments(line); const out = []; for (const m of masked.matchAll(BOLD_PUBLISHED)) { const at = m.index + m[0].length; out.push({ kind: "bold", value: m[1].replace(/[\s,%]/g, ""), text: m[0].trim(), at, numAt: m.index + m[0].indexOf(m[1]), marked: MARKED_AFTER.test(line.slice(at)) }); } return out; } /* The marker check, deliberately looser than check-cited's CITE. That reader is strict on purpose; here, ANY citation-shaped comment beside the figure means somebody marked it and check-cited owns the question of whether they marked it correctly. Two guards disagreeing about what counts as marked would leave a figure that neither of them holds. */ const MARKED_AFTER = /^[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*` instead. It must give a reason: an empty waiver is a decision nobody recorded, and it is fatal — the call made for the empty cast marker in R-288. Every waiver prints on every green build, so choosing to stop checking something stays visible instead of becoming the default. */ const WAIVED_AFTER = /^[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*/i; const waiverAt = rest => { const m = WAIVED_AFTER.exec(rest); return m ? { reason: m[1] } : null; }; /** Every measurement-shaped figure on one line, with whether it carries a marker. */ export function figuresIn(line) { const out = []; const seen = new Set(); for (const [kind, re] of SHAPES) { re.lastIndex = 0; let m; while ((m = re.exec(line))) { /* THE CAPTURE'S OWN OFFSET, NOT A SEARCH FOR ITS TEXT. `line.indexOf(m[1], m.index)` finds the FIRST copy of the digits at or after the match start, which is not necessarily the one that was captured: "the wipe rate of 74 in ten is 74%" matches with m[1] = "74", and the search lands on the earlier 74. numEnd then points into the middle of the sentence, the marker test reads " in ten is 74%…" and a correctly cited figure is reported as bare. Found by a stranger's read of the fold; it needs an integer duplicate inside the one shape with a wide gap before its capture, so it is contrived — but its failure is the bad direction, a FALSE POSITIVE on good prose, which is the class that teaches a reader to stop reading the output. The `d` flag gives the capture's real position and there is nothing to search for. */ const numEnd = m.indices[1][1]; /* Context, not the match: the word that excuses a figure sits outside it. */ const before = line.slice(Math.max(0, m.index - 24), numEnd); if (NOT_A_MEASUREMENT.some(([re2]) => re2.test(before) || re2.test(m[0]))) continue; /* One figure can match two shapes ("2 rounds" inside "median 2 rounds"); the figure is the number, so it is counted once at the position it occupies. */ if (seen.has(numEnd)) continue; seen.add(numEnd); out.push({ kind, value: m[1], text: m[0].trim(), at: numEnd, numAt: numEnd - m[1].length, marked: MARKED_AFTER.test(line.slice(numEnd)) }); } } return out.sort((a, b) => a.text.localeCompare(b.text)); } /** * Every figure in one document, by all three readers. Exported and taking a STRING because * a reader tested by editing the corpus proves the corpus, and goes quiet the day somebody * rewords the sentence it was anchored to (R-292). */ export function scanDocument(label, src) { const lines = src.split("\n"); /* Fenced blocks are the safe harbour, exactly as they are for check-outcomes: a code sample or a recorded harness dump is not the document making a claim. */ const fenced = new Set(); let inFence = false; lines.forEach((ln, i) => { if (/^\s*```/.test(ln)) { inFence = !inFence; fenced.add(i); return; } if (inFence) fenced.add(i); }); const optedIn = /