From 0de463c470d3f0eeda351e98db50725bbce4dcf9 Mon Sep 17 00:00:00 2001 From: slaguru666 <111923774+slaguru666@users.noreply.github.com> Date: Sun, 13 Sep 2026 10:29:45 +0100 Subject: [PATCH] check-cited: make the regex match its own error message, and name the right lines MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two residual defects in the marker check added at 4d7d93a, both found by the peer session that found the original. 1. THE REGEX WAS LOOSER THAN ITS DOCUMENTATION. The failure text has always promised the comment must follow its figure "immediately, with nothing but markup between", but CITE's separators were \s*, which spans newlines — so a marker on the line after its number resolved and was checked, contrary to the stated rule. Tightened to [ \t] throughout. Verified safe before changing it: matched under the loose pattern 25, under the tight pattern 25, so no citation in CLEAN GROUND relies on crossing a newline. There is no finding in the document. 2. THE OFFENDER REPORT WAS RIGHT ABOUT THE COUNT AND WRONG ABOUT THE LINES. It filtered line by line, so a valid cross-line citation was printed as an offender whenever some other marker was genuinely unread — a maintainer told "line 15 is broken" would have edited a working citation. That is the headline defect fixed an hour ago one level down: correct verdict, wrong reason. Offenders are now located by position across the whole text, so the lines named are exactly the markers no CITE match covers. This is the fix that matters independently of the regex: a marker spanning lines would still be counted once by ANY_CITE and named nowhere, so tightening alone leaves a narrower version of the same bug, and position-based reporting stays correct if anyone ever loosens the separator again. Placement battery, exit codes read directly: clean tree OK 25 no false positive; one word between number and marker fails; bare marker alone fails; two on a line with one attached fails naming only the unattached; cross-line now fails rather than silently passing. Three iterations of this guard, three variants of one defect — a marker or a message that exists and is not read — and all three were found by the session that did not write it. npm run check: 16 guards, exit 0. Co-Authored-By: Claude Opus 5 --- tools/check-cited.mjs | 33 ++++++++++++++++++++++----------- 1 file changed, 22 insertions(+), 11 deletions(-) diff --git a/tools/check-cited.mjs b/tools/check-cited.mjs index bfe42e1..90cd117 100644 --- a/tools/check-cited.mjs +++ b/tools/check-cited.mjs @@ -70,8 +70,13 @@ function resolve(json, dotted) { } /* The number immediately before the marker, allowing for markup and a trailing unit: - "**40**", "25.5%", "1.4x". Anchored to the end so it is the nearest one. */ -const CITE = /([\d]+(?:\.[\d]+)?)\s*(?:%|x)?\**\s*(?:—|-|–)?\s*/g; + "**40**", "25.5%", "1.4x". Anchored to the end so it is the nearest one. + [ \t] rather than \s throughout, so a citation cannot span a newline. The failure text + below has always promised the comment must follow its figure "immediately, with nothing + but markup between"; \s* quietly allowed a marker on the following line, so the code was + looser than its own documentation. Making them agree also keeps the offender report + honest — see the count check. */ +const CITE = /([\d]+(?:\.[\d]+)?)[ \t]*(?:%|x)?\**[ \t]*(?:—|-|–)?[ \t]*/g; /* Every marker, however badly placed. The two counts must agree — see below. */ const ANY_CITE = //g; @@ -89,18 +94,24 @@ for (const [rel, abs] of scenarioFiles()) { number; put one word between them and the citation is silently skipped while the run still reports OK with the same count as a clean tree. Found by a peer session writing a bad citation by accident. So every occurrence is counted and the totals must agree. */ - const seen = (text.match(ANY_CITE) ?? []).length; - const read = [...text.matchAll(CITE)].length; - if (seen !== read) { - const unread = text.split("\n") - .map((l, i) => [i + 1, l]) - .filter(([, l]) => ANY_CITE.test(l) && (ANY_CITE.lastIndex = 0) === 0 - && [...l.matchAll(CITE)].length < (l.match(ANY_CITE) ?? []).length); - problems.push(` ${rel}: ${seen - read} citation marker(s) not attached to a number, so ` + const markers = [...text.matchAll(ANY_CITE)]; + const spans = [...text.matchAll(CITE)].map(m => [m.index, m.index + m[0].length]); + /* Located by position, not per line: a valid citation is a marker some CITE match covers. + The first version of this filtered line by line, which named any VALID citation that + happened to share a line with nothing — reporting a working marker as broken whenever + another one was genuinely unread. The count was right and the explanation was wrong, + which is the defect this guard exists to prevent, two levels down. */ + const unread = markers.filter(m => !spans.some(([a, b]) => m.index >= a && m.index < b)); + if (unread.length) { + const lineOf = i => text.slice(0, i).split("\n").length; + problems.push(` ${rel}: ${unread.length} citation marker(s) not attached to a number, so ` + `nothing checks them — a marker that looks like a citation and is never read is the ` + `defect this guard exists for. The comment must follow its figure immediately ` + `(\`**40**\`), with nothing but markup between.`); - unread.forEach(([n, l]) => problems.push(` line ${n}: ${l.trim().slice(0, 110)}`)); + unread.forEach(m => { + const n = lineOf(m.index); + problems.push(` line ${n}: ${text.split("\n")[n - 1].trim().slice(0, 110)}`); + }); } for (const m of text.matchAll(CITE)) {