From bd9b3bdd2777470bdb3c97ae27694b94f39c245a Mon Sep 17 00:00:00 2001 From: slaguru666 <111923774+slaguru666@users.noreply.github.com> Date: Sun, 13 Sep 2026 20:35:43 +0100 Subject: [PATCH] check-figures: use the capture's position, not a search for its text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit R-308. Read the fold as a stranger at its author's request. Two of the three risks they flagged are sound: isProse drops nothing that carries a citation (scanned the whole corpus), and the duration exclusion earns its place, since "runs" now means three things in this corpus and only the cell can tell them apart. The third is real. numEnd used line.indexOf(m[1], m.index), which finds the first copy of the digits at or after the match start rather than the copy that was captured: "The wipe rate of 74 in ten is 74%." value=74 numAt=17 marked=FALSE A correctly cited figure reported bare, because numEnd lands mid-sentence and the marker test reads " in ten is 74%...". Fixed with the d flag: m.indices[1] gives the capture's real position and there is nothing to search for. Narrow to reach — it needs the wipe-rate shape, the only one with a wide gap before its capture, and an integer duplicate inside that gap; a decimal cannot do it because [^.] will not span a decimal point. Three attempts failed for that reason before the fourth worked, which is why this is recorded as narrow rather than theoretical. Worth fixing anyway because its direction is the bad one: a miss costs one figure, a false positive on correct prose costs the guard. Seven-case battery re-run, every restore byte-identical. Co-Authored-By: Claude Opus 5 --- docs/REVIEW_LOG.md | 63 +++++++++++++++++++++++++++++++++++++++++ tools/check-figures.mjs | 29 +++++++++++++------ 2 files changed, 83 insertions(+), 9 deletions(-) diff --git a/docs/REVIEW_LOG.md b/docs/REVIEW_LOG.md index f32b5eb..1996e6b 100644 --- a/docs/REVIEW_LOG.md +++ b/docs/REVIEW_LOG.md @@ -7728,3 +7728,66 @@ stop checking something stays visible rather than becoming the default. Six toda **21 guards, 107 behavioural tests, 139 figures with 115 marked and 6 waived.** The reader tests moved onto `scanDocument` and run on strings, so the shapes they name survive the file that no longer exists. + +--- + +## R-308 — the stranger's read: one real defect in the fold, and two risks that were not + +**Read as a stranger at the fold's request**, looking hardest at the three places its author +named. Two of the three are sound; the third is a genuine false-positive generator and is +fixed here. + +**`isProse` — sound.** The eight-word threshold is a guess and its author said so, the worry +being that a long legitimate cell would be dropped in silence. Scanned every measurement column +in the corpus for a cell that `isProse` excludes *and* that carries a citation: **none.** +Nothing currently held is being dropped. The guess is still a guess and the test above is the +one to re-run if a table ever grows a wordier cell. + +**The duration list — sound, and it earns its place.** *Runs* now means three different things +in one corpus: a sample count (*"Measured over 300 runs"*), a percentile phrase (*"1 in 100 +runs past"*), and a duration (*"Runs | ~10 min"*). Only the middle one is a measurement. No +word list can separate three senses of one word; the cell test does. + +### ⚠ The defect: `numEnd` searched for the capture's text instead of using its position + +```js +const numEnd = line.indexOf(m[1], m.index) + m[1].length; +``` + +`indexOf` finds the **first** copy of those digits at or after the match start, which need not +be the copy that was captured. + +``` +"The wipe rate of 74 in ten is 74%." + wipe-rate value=74 numAt=17 marked=FALSE <- points at the 74 in "of 74" +``` + +**The figure is correctly cited and is reported bare.** `numEnd` lands mid-sentence, the marker +test reads *" in ten is 74%…"* and finds nothing. + +**Reachability is narrow and was established by trying to break it four ways rather than by +reading.** It needs the `wipe-rate` shape — the only one with a wide gap before its capture — +and an **integer** duplicate inside that gap. A *decimal* duplicate cannot do it, because the +shape's `[^.]{0,24}?` will not span a decimal point. Three contrived attempts failed for that +reason before the fourth succeeded, and each failure was informative: **the near-misses are why +this is recorded as narrow rather than as theoretical.** + +**Fixed with the `d` flag**, so each match reports where its capture actually sits and there is +nothing to search for: + +```js +const numEnd = m.indices[1][1]; +``` + +⚠ **Why a contrived false positive was worth fixing at all.** Its direction is the bad one. A +miss costs one wrong figure; a false positive on correct prose costs the guard, because what a +reader learns from output that flags good material is to stop reading the output. That is R-301's +lesson — 130 false positives off citation-marker digits — arriving from a different direction, +and it is the second time in this file that the same judgement has decided a fix. + +**Battery re-run after the change**, seven cases, every restore byte-identical: bold prose +stripped fails, unbolded cells fail, the historical bare `24` fails, a planted *"median 24 +rounds"* fails, the ratchet trips, **the cited-figure-with-duplicate case is now silent**, and +the clean tree is silent. + +**State:** 21 guards, one merged reader, CLEAN GROUND v0.24. diff --git a/tools/check-figures.mjs b/tools/check-figures.mjs index 8ae9609..c6cc718 100644 --- a/tools/check-figures.mjs +++ b/tools/check-figures.mjs @@ -67,15 +67,16 @@ const N = String.raw`\*{0,2}(\d+(?:\.\d+)?)\*{0,2}`; steps", "an engineer on his rounds", "01:06". A guard that cries wolf 282 times is a guard nobody runs. These are the phrasings the documents actually use when quoting the simulator, and each one was read off a figure that IS cited somewhere. */ +/* `d` (hasIndices) so each match reports where its capture actually sits — see numEnd. */ const SHAPES = [ - ["median", new RegExp(String.raw`median(?:\s+of)?\s+` + N, "gi")], - ["rounds", new RegExp(N + String.raw`\s*\*{0,2}\s*rounds\b`, "gi")], - ["wiped", new RegExp(N + String.raw`\s*%?\s*\*{0,2}\s*(?:wiped|wipes)\b`, "gi")], - ["wipes-pct", new RegExp(String.raw`wipe[sd]?\s+` + N + String.raw`\s*%`, "gi")], - ["wipe-rate", new RegExp(String.raw`wipe rate[^.]{0,24}?` + N + String.raw`\s*%`, "gi")], - ["of-N-hurt", new RegExp(N + String.raw`\s*\*{0,2}\s*of\s+\d+\s+(?:hurt|down)\b`, "gi")], - ["deaths", new RegExp(N + String.raw`\s*\*{0,2}\s*deaths?\b`, "gi")], - ["pct-of-runs", new RegExp(N + String.raw`\s*%[^.]{0,18}\bruns?\b`, "gi")] + ["median", new RegExp(String.raw`median(?:\s+of)?\s+` + N, "gid")], + ["rounds", new RegExp(N + String.raw`\s*\*{0,2}\s*rounds\b`, "gid")], + ["wiped", new RegExp(N + String.raw`\s*%?\s*\*{0,2}\s*(?:wiped|wipes)\b`, "gid")], + ["wipes-pct", new RegExp(String.raw`wipe[sd]?\s+` + N + String.raw`\s*%`, "gid")], + ["wipe-rate", new RegExp(String.raw`wipe rate[^.]{0,24}?` + N + String.raw`\s*%`, "gid")], + ["of-N-hurt", new RegExp(N + String.raw`\s*\*{0,2}\s*of\s+\d+\s+(?:hurt|down)\b`, "gid")], + ["deaths", new RegExp(N + String.raw`\s*\*{0,2}\s*deaths?\b`, "gid")], + ["pct-of-runs", new RegExp(N + String.raw`\s*%[^.]{0,18}\bruns?\b`, "gid")] ]; /* WHAT IS NOT A MEASUREMENT, each with the case that put it here. @@ -274,7 +275,17 @@ export function figuresIn(line) { re.lastIndex = 0; let m; while ((m = re.exec(line))) { - const numEnd = line.indexOf(m[1], m.index) + m[1].length; + /* THE CAPTURE'S OWN OFFSET, NOT A SEARCH FOR ITS TEXT. `line.indexOf(m[1], m.index)` + finds the FIRST copy of the digits at or after the match start, which is not + necessarily the one that was captured: "the wipe rate of 74 in ten is 74%" matches + with m[1] = "74", and the search lands on the earlier 74. numEnd then points into + the middle of the sentence, the marker test reads " in ten is 74%…" and a correctly + cited figure is reported as bare. Found by a stranger's read of the fold; it needs + an integer duplicate inside the one shape with a wide gap before its capture, so it + is contrived — but its failure is the bad direction, a FALSE POSITIVE on good prose, + which is the class that teaches a reader to stop reading the output. + The `d` flag gives the capture's real position and there is nothing to search for. */ + const numEnd = m.indices[1][1]; /* Context, not the match: the word that excuses a figure sits outside it. */ const before = line.slice(Math.max(0, m.index - 24), numEnd); if (NOT_A_MEASUREMENT.some(([re2]) => re2.test(before) || re2.test(m[0]))) continue;