Fourth variant of the same defect, found by the peer session probing sideways.
Neither CITE nor ANY_CITE carried the `i` flag, and ANY_CITE required a literal
"cite:" with no space before the colon. So three plausible spellings were
invisible to the reader AND to the counter that exists to catch invisible
markers:
<!-- Cite: first-blood cut.swing --> OK 25, exit 0
<!-- CITE: first-blood cut.swing --> OK 25, exit 0
<!-- cite : first-blood cut.swing --> OK 25, exit 0
Each of those was verified carrying a citation printing 41 against an artifact
holding 40, and each reported OK with a count byte-identical to a clean tree.
The count check could not see them because it was looking for the same literal
the reader was. Capitalising the first word of a comment is not an exotic
mistake.
The two patterns are now deliberately asymmetric, which is the actual fix:
CITE tolerates case and spaces around the colon, so those spellings
simply work when correctly placed.
ANY_CITE stays looser still, so a spelling neither of us anticipated is
counted and named rather than skipped.
The reader accepts only what the format specifies; the counter recognises
anything a person might have meant as a citation; the difference is reported.
The failure text now covers misspelling as well as misplacement, since the
unread set can be either.
Matrix verified, exit codes read directly: a wrong value fails under all four
spellings including no-spaces; a right value passes under all of them, counting
26; a misplaced-and-capitalised marker fails; a marker missing its path fails;
clean tree OK 25 with no false positive.
Four iterations of this guard, four variants of one defect -- something that
exists and is never read -- and all four were found by the session that did not
write it.
npm run check: 16 guards, exit 0.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
187 lines
9.3 KiB
JavaScript
187 lines
9.3 KiB
JavaScript
/**
|
||
* A figure a scenario prints must still be the figure its artifact holds.
|
||
*
|
||
* check-firstblood and check-attackers both compare a baseline against a fresh run, so
|
||
* they catch the GAME changing. Neither of them reads the document. Re-record after a
|
||
* re-cast and the artifact updates, the guard goes green, and the paragraph goes on
|
||
* printing the old number with a citation under it saying where the new one lives. The
|
||
* citation makes it worse, not better: it tells a GM the figure is checked.
|
||
*
|
||
* So the citations are machine-readable. A figure copied out of an artifact is written
|
||
*
|
||
* a swing of **40**<!-- cite: first-blood cut.swing -->
|
||
*
|
||
* and this guard resolves every one of them against the named JSON. Copy a number wrong,
|
||
* or re-record and leave the prose alone, and the build stops with both values named.
|
||
*
|
||
* This is the check-bestiary arrangement one size down. That document is GENERATED from
|
||
* the game, so it cannot drift; a scenario is written by hand and cannot be, but the
|
||
* numbers inside it can be held to the same standard.
|
||
*
|
||
* node tools/check-cited.mjs check every citation in every scenario
|
||
* node tools/check-cited.mjs --list print what each one currently resolves to
|
||
*/
|
||
import { readFileSync, existsSync } from "node:fs";
|
||
import path from "node:path";
|
||
import { scenarioFiles } from "./check-scenarios.mjs";
|
||
|
||
const ROOT = path.resolve(path.dirname(new URL(import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1")), "..");
|
||
const LIST = process.argv.slice(2).includes("--list");
|
||
|
||
/* Artifact name as written in a citation -> the file it means. Adding a baseline here is
|
||
what makes it citeable; nothing else in the document needs to know. */
|
||
const ARTIFACTS = {
|
||
"first-blood": "tools/first-blood-baseline.json",
|
||
"attackers": "tools/attackers-baseline.json",
|
||
"fight-tail": "tools/fight-tail-baseline.json",
|
||
"lethality": "tools/lethality-baseline.json"
|
||
};
|
||
|
||
/* Fields that exist in an artifact but must never appear in prose. A sample maximum is
|
||
the clearest case: same party, same seeds, same runs, and renaming a config moved
|
||
fight-tail's `longest` from 71 to 90 rounds, because the stream is derived from the id.
|
||
It reads like a bound on the encounter and is a property of the label. Citing it would
|
||
be the n=1 mistake one level up, so the citation itself is refused. */
|
||
const UNCITEABLE = {
|
||
longest: "a sample maximum, not a bound — it moves by a third of its value on a re-seed. "
|
||
+ "Cite p99 instead: 'one fight in a hundred runs past N rounds' is stable to about three rounds."
|
||
};
|
||
|
||
const loaded = new Map();
|
||
function artifact(name) {
|
||
if (loaded.has(name)) return loaded.get(name);
|
||
const rel = ARTIFACTS[name];
|
||
if (!rel) return null;
|
||
const file = path.join(ROOT, rel);
|
||
const json = existsSync(file) ? JSON.parse(readFileSync(file, "utf8")) : null;
|
||
loaded.set(name, json);
|
||
return json;
|
||
}
|
||
|
||
/** "configs.cut.swing" or "cut.swing" — the leading container is optional. */
|
||
function resolve(json, dotted) {
|
||
const direct = dotted.split(".").reduce((o, k) => (o == null ? o : o[k]), json);
|
||
if (direct !== undefined) return direct;
|
||
for (const container of ["configs", "agents"]) {
|
||
const v = dotted.split(".").reduce((o, k) => (o == null ? o : o[k]), json?.[container]);
|
||
if (v !== undefined) return v;
|
||
}
|
||
return undefined;
|
||
}
|
||
|
||
/* The number immediately before the marker, allowing for markup and a trailing unit:
|
||
"**40**", "25.5%", "1.4x". Anchored to the end so it is the nearest one.
|
||
[ \t] rather than \s throughout, so a citation cannot span a newline. The failure text
|
||
below has always promised the comment must follow its figure "immediately, with nothing
|
||
but markup between"; \s* quietly allowed a marker on the following line, so the code was
|
||
looser than its own documentation. Making them agree also keeps the offender report
|
||
honest — see the count check. */
|
||
const CITE = /([\d]+(?:\.[\d]+)?)[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*<!--[ \t]*cite[ \t]*:[ \t]*([\w-]+)[ \t]+([\w.]+)[ \t]*-->/gi;
|
||
|
||
/* ANY marker a person might have meant as a citation, however spelled or placed. The two
|
||
patterns are deliberately ASYMMETRIC: the reader accepts only what the format specifies,
|
||
the counter recognises anything citation-shaped, and the difference is reported.
|
||
|
||
Both were once the same literal, and that made the count check blind in exactly the way
|
||
it existed to prevent — neither carried `i`, so `<!-- Cite: ... -->`, `<!-- CITE: ... -->`
|
||
and `<!-- cite : ... -->` were invisible to the reader AND to the counter watching for
|
||
invisible markers. A citation printing 41 against an artifact holding 40 passed under all
|
||
three, reporting OK with a count identical to a clean tree. Capitalising the first word of
|
||
a comment is not an exotic mistake; it is what a person does without thinking.
|
||
|
||
CITE now tolerates case and a space around the colon, so those spellings simply work.
|
||
ANY_CITE stays looser still, so a spelling neither of us anticipated is counted and
|
||
named rather than skipped. */
|
||
const ANY_CITE = /<!--[ \t]*cite[ \t]*:[^>]*-->/gi;
|
||
|
||
let checked = 0;
|
||
const problems = [];
|
||
const rows = [];
|
||
|
||
/* Deliberately NOT scenarioText(): that strips HTML comments, which is where the
|
||
citations live. The raw file is the thing being checked. */
|
||
for (const [rel, abs] of scenarioFiles()) {
|
||
const text = readFileSync(abs, "utf8");
|
||
|
||
/* A MARKER THIS GUARD CANNOT READ IS WORSE THAN NO MARKER AT ALL, because whoever wrote
|
||
it believes they added a check. CITE requires the comment to sit immediately after its
|
||
number; put one word between them and the citation is silently skipped while the run
|
||
still reports OK with the same count as a clean tree. Found by a peer session writing
|
||
a bad citation by accident. So every occurrence is counted and the totals must agree. */
|
||
const markers = [...text.matchAll(ANY_CITE)];
|
||
const spans = [...text.matchAll(CITE)].map(m => [m.index, m.index + m[0].length]);
|
||
/* Located by position, not per line: a valid citation is a marker some CITE match covers.
|
||
The first version of this filtered line by line, which named any VALID citation that
|
||
happened to share a line with nothing — reporting a working marker as broken whenever
|
||
another one was genuinely unread. The count was right and the explanation was wrong,
|
||
which is the defect this guard exists to prevent, two levels down. */
|
||
const unread = markers.filter(m => !spans.some(([a, b]) => m.index >= a && m.index < b));
|
||
if (unread.length) {
|
||
const lineOf = i => text.slice(0, i).split("\n").length;
|
||
problems.push(` ${rel}: ${unread.length} citation marker(s) this guard cannot read, so `
|
||
+ `nothing checks them — a marker that looks like a citation and is never read is the `
|
||
+ `defect this guard exists for. Either it is misplaced (the comment must follow its `
|
||
+ `figure immediately, \`**40**<!-- cite: artifact path -->\`, on the same line with `
|
||
+ `nothing but markup between) or it is misspelled (case and spaces around the colon `
|
||
+ `are fine; the shape \`cite: <artifact> <dotted.path>\` is not optional).`);
|
||
unread.forEach(m => {
|
||
const n = lineOf(m.index);
|
||
problems.push(` line ${n}: ${text.split("\n")[n - 1].trim().slice(0, 110)}`);
|
||
});
|
||
}
|
||
|
||
for (const m of text.matchAll(CITE)) {
|
||
const [, printed, name, dotted] = m;
|
||
checked++;
|
||
const json = artifact(name);
|
||
if (!json) {
|
||
problems.push(` ${rel}: cites "${name}", which is not a known artifact `
|
||
+ `(${Object.keys(ARTIFACTS).join(", ")})`);
|
||
continue;
|
||
}
|
||
const held = resolve(json, dotted);
|
||
if (held === undefined) {
|
||
problems.push(` ${rel}: cites ${name} ${dotted}, which that artifact does not hold`);
|
||
continue;
|
||
}
|
||
const leaf = dotted.split(".").pop();
|
||
if (UNCITEABLE[leaf]) {
|
||
problems.push(` ${rel}: cites ${name} ${dotted}, which must not be quoted — ${UNCITEABLE[leaf]}`);
|
||
continue;
|
||
}
|
||
rows.push([rel, `${name} ${dotted}`, printed, String(held)]);
|
||
if (Number(printed) !== Number(held)) {
|
||
problems.push(` ${rel}: prints ${printed} but ${ARTIFACTS[name]} holds ${held} `
|
||
+ `(${dotted}). Re-recording updated the artifact and left the prose behind — fix the sentence.`);
|
||
}
|
||
}
|
||
}
|
||
|
||
if (LIST) {
|
||
for (const [f, src, printed, held] of rows) {
|
||
console.log(` ${f.padEnd(32)} ${src.padEnd(28)} prints ${printed.padStart(6)} holds ${held}`);
|
||
}
|
||
process.exit(0);
|
||
}
|
||
|
||
/* A guard that passes because it found nothing to check is the defect this suite has now
|
||
been caught by four times, so it is not allowed here: the citations exist, and a run
|
||
that cannot see them means the markers were reformatted away, not that the prose is
|
||
clean. */
|
||
if (!checked) {
|
||
console.error("check-cited: FAILED — no cited figures found at all. The scenarios carry "
|
||
+ "`<!-- cite: artifact path -->` markers next to every measured number; finding none "
|
||
+ "means they have been stripped or reformatted, not that there is nothing to check.");
|
||
process.exit(1);
|
||
}
|
||
|
||
if (problems.length) {
|
||
console.error("check-cited: FAILED — a scenario's cited figures do not match their artifacts");
|
||
problems.forEach(p => console.error(p));
|
||
process.exit(1);
|
||
}
|
||
|
||
const artifacts = new Set(rows.map(r => r[1].split(" ")[0]));
|
||
console.log(`check-cited: OK — ${checked} cited figures across ${scenarioFiles().length} scenario files `
|
||
+ `resolve against ${artifacts.size} baselines`);
|