diff --git a/tools/check-cited.mjs b/tools/check-cited.mjs index bfe42e1..90cd117 100644 --- a/tools/check-cited.mjs +++ b/tools/check-cited.mjs @@ -70,8 +70,13 @@ function resolve(json, dotted) { } /* The number immediately before the marker, allowing for markup and a trailing unit: - "**40**", "25.5%", "1.4x". Anchored to the end so it is the nearest one. */ -const CITE = /([\d]+(?:\.[\d]+)?)\s*(?:%|x)?\**\s*(?:—|-|–)?\s*/g; + "**40**", "25.5%", "1.4x". Anchored to the end so it is the nearest one. + [ \t] rather than \s throughout, so a citation cannot span a newline. The failure text + below has always promised the comment must follow its figure "immediately, with nothing + but markup between"; \s* quietly allowed a marker on the following line, so the code was + looser than its own documentation. Making them agree also keeps the offender report + honest — see the count check. */ +const CITE = /([\d]+(?:\.[\d]+)?)[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*/g; /* Every marker, however badly placed. The two counts must agree — see below. */ const ANY_CITE = //g; @@ -89,18 +94,24 @@ for (const [rel, abs] of scenarioFiles()) { number; put one word between them and the citation is silently skipped while the run still reports OK with the same count as a clean tree. Found by a peer session writing a bad citation by accident. So every occurrence is counted and the totals must agree. */ - const seen = (text.match(ANY_CITE) ?? []).length; - const read = [...text.matchAll(CITE)].length; - if (seen !== read) { - const unread = text.split("\n") - .map((l, i) => [i + 1, l]) - .filter(([, l]) => ANY_CITE.test(l) && (ANY_CITE.lastIndex = 0) === 0 - && [...l.matchAll(CITE)].length < (l.match(ANY_CITE) ?? []).length); - problems.push(` ${rel}: ${seen - read} citation marker(s) not attached to a number, so ` + const markers = [...text.matchAll(ANY_CITE)]; + const spans = [...text.matchAll(CITE)].map(m => [m.index, m.index + m[0].length]); + /* Located by position, not per line: a valid citation is a marker some CITE match covers. + The first version of this filtered line by line, which named any VALID citation that + happened to share a line with nothing — reporting a working marker as broken whenever + another one was genuinely unread. The count was right and the explanation was wrong, + which is the defect this guard exists to prevent, two levels down. */ + const unread = markers.filter(m => !spans.some(([a, b]) => m.index >= a && m.index < b)); + if (unread.length) { + const lineOf = i => text.slice(0, i).split("\n").length; + problems.push(` ${rel}: ${unread.length} citation marker(s) not attached to a number, so ` + `nothing checks them — a marker that looks like a citation and is never read is the ` + `defect this guard exists for. The comment must follow its figure immediately ` + `(\`**40**\`), with nothing but markup between.`); - unread.forEach(([n, l]) => problems.push(` line ${n}: ${l.trim().slice(0, 110)}`)); + unread.forEach(m => { + const n = lineOf(m.index); + problems.push(` line ${n}: ${text.split("\n")[n - 1].trim().slice(0, 110)}`); + }); } for (const m of text.matchAll(CITE)) {