collectSpecs() in tools/all-specs.mjs is the single place that decides which specs exist. Nine modules were importing NPCS and PREGENS straight from content.mjs instead, so each carried its own idea of the population and measured a different subset of the game. check-lethality was the clearest case: it scored 47 creatures and reported OK for all 184. check-seam (guard 24, first in the suite) scans every module and fails if anything but all-specs.mjs names NPCS or PREGENS in a content.mjs import. The nine violators are re-pointed at the seam. Re-recording check-focus's baseline against the full 184 raised it from 47 packs to 138 and surfaced one creature focus fire does not help: the dun cow. Rather than re-record that away, the guard now requires the advice section of BESTIARY.md to name every such exception, and bestiary.mjs generates the sentence. The first version of that check asked whether the name appeared anywhere in BESTIARY.md, which every creature's own heading satisfies — deleting the exception sentence still passed. It reads only the "Shooting at something that moves" section now, and the mutation test fails as it should. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
281 lines
14 KiB
JavaScript
281 lines
14 KiB
JavaScript
/**
|
|
* fight-tail — how long these fights run, and what a median hides.
|
|
*
|
|
* Fourteen guards measured this scenario's combat and not one could have told a GM that a
|
|
* twenty-round fight was ordinary, because every one of them averages. EXPOSURE prints
|
|
* median rounds, and a median is the least useful number for a convention slot: the GM is
|
|
* not planning for the typical fight, they are planning for the one that eats the act.
|
|
*
|
|
* node tools/fight-tail.mjs read the recorded distribution
|
|
* node tools/fight-tail.mjs --check compare against the baseline
|
|
* node tools/fight-tail.mjs --update re-record it
|
|
*
|
|
* THE CAP IS THE MEASUREMENT HERE, which is why this tool passes its own.
|
|
*
|
|
* runFight stops at maxRounds and scores the result by who is standing, so a fight that
|
|
* would have run longer is recorded as exactly maxRounds and counts as neither a win nor a
|
|
* wipe. At the 40-round default that censors the top of the distribution — the six-a-side
|
|
* 99th percentile reads 40 because 40 is the wall, not because the fights end there. For
|
|
* every other guard that is a rounding error on a mean. For this one it is the finding, so
|
|
* the cap is set far past any observed fight and the check refuses to record a baseline if
|
|
* anything reaches it. R-271 documented the censoring; this tool is the one that cannot
|
|
* tolerate it.
|
|
*
|
|
* TWO CLAIMS, because the obvious one does not discriminate. "The long end is roughly twice
|
|
* the median" is true of both encounters and so tells a GM nothing about which is which.
|
|
* What is worth knowing is that LENGTH PREDICTS DEATH — a fight still going at fifteen
|
|
* rounds is not a stalemate, it is one the party is losing slowly — and that the SIX-A-SIDE
|
|
* fight is the longer one, which is counterintuitive until R-270: more bodies means more of
|
|
* them fighting on at reduced skill instead of dropping. The cut is shorter because it is
|
|
* decisive, not because it is safer.
|
|
*
|
|
* "Past N rounds" means strictly more than N. Stated because a >= count reads five points
|
|
* higher at N=15 and the two are easy to confuse in prose.
|
|
*/
|
|
import { readFile, writeFile } from "node:fs/promises";
|
|
import { existsSync } from "node:fs";
|
|
import path from "node:path";
|
|
/* The whole population, through the one door. Reading NPCS from content.mjs meant
|
|
this tool could only see the 47 creatures that happen to live in that file, so a
|
|
creature from tools/bestiary-*.mjs was simply "not found". See check-seam.mjs. */
|
|
import { collectSpecs } from "./all-specs.mjs";
|
|
const NPCS = (await collectSpecs()).map(r => r.spec);
|
|
import { ROSTER } from "./roster.mjs";
|
|
import { runFight, seedFor, makeRng } from "./simulate.mjs";
|
|
import { castAndCut } from "./declared-cast.mjs";
|
|
|
|
const ROOT = path.resolve(path.dirname(new URL(import.meta.url).pathname.replace(/^\/([A-Za-z]:)/, "$1")), "..");
|
|
const BASELINE = path.join(ROOT, "tools", "fight-tail-baseline.json");
|
|
|
|
const argv = process.argv.slice(2);
|
|
const UPDATE = argv.includes("--update");
|
|
const CHECK = argv.includes("--check");
|
|
|
|
const RUNS = 2000;
|
|
const SEEDS = [11, 4242, 90210];
|
|
/* Far past the longest fight ever observed here (71 rounds). Not runFight's default of 40,
|
|
which truncates precisely what this tool measures. */
|
|
const CAP = 400;
|
|
/* The round by which a convention act is gone whatever the party size, so the long/short
|
|
split is an absolute number rather than a per-config percentile. */
|
|
const LONG = 15;
|
|
|
|
const { line: LINE, cut: CUT } = castAndCut("fight-tail");
|
|
const CONFIGS = [
|
|
{ id: "cut", label: "four of the roster vs three of the column", party: CUT, creature: "quiet_neighbours_npc", count: 3 },
|
|
{ id: "column", label: "six of the roster vs six of the column", party: LINE, creature: "quiet_neighbours_npc", count: 6 }
|
|
];
|
|
|
|
/* How much likelier a long fight must be to end in a wipe before the claim counts as true.
|
|
Measured at 1.37x in the cut (seed noise 0.1) and 2.42x at six a side (seed noise 1.1 —
|
|
forty-five per cent of the value, which is why no prose quotes that ratio). The bar sits
|
|
at roughly twice the cut's noise below the cut's figure, so it asserts the SIGN of the
|
|
relationship and nothing finer: the point is to catch it disappearing, not to pin a
|
|
number three seeds cannot support. */
|
|
const DEADLIER = 1.15;
|
|
|
|
const roster = keys => keys.map(k => {
|
|
const found = ROSTER.find(r => r.key === k || r.key === "pc_" + k);
|
|
if (!found) { console.error(`fight-tail: "${k}" is not on the duty roster`); process.exit(1); }
|
|
return found;
|
|
});
|
|
const creature = key => {
|
|
const spec = NPCS.find(n => n.key === key);
|
|
if (!spec) { console.error(`fight-tail: no creature keyed "${key}"`); process.exit(1); }
|
|
return spec;
|
|
};
|
|
|
|
const mean = a => a.reduce((x, y) => x + y, 0) / a.length;
|
|
const r1 = n => Number(n.toFixed(1));
|
|
|
|
/** One sweep. Lengths are kept whole rather than averaged, which is the entire point. */
|
|
function sweep(party, foes, seed, runs) {
|
|
const rounds = [];
|
|
let capped = 0, undecided = 0;
|
|
const longF = { n: 0, wiped: 0 }, shortF = { n: 0, wiped: 0 };
|
|
for (let i = 0; i < runs; i++) {
|
|
const r = runFight(makeRng(seed + i * 2654435761), party, foes, { maxRounds: CAP });
|
|
rounds.push(r.rounds);
|
|
if (r.rounds >= CAP) capped++;
|
|
if (!r.wiped && !r.won) undecided++;
|
|
const bucket = r.rounds > LONG ? longF : shortF;
|
|
bucket.n++;
|
|
if (r.wiped) bucket.wiped++;
|
|
}
|
|
rounds.sort((a, b) => a - b);
|
|
const q = p => rounds[Math.min(rounds.length - 1, Math.floor(p * rounds.length))];
|
|
const over = n => (100 * rounds.filter(x => x > n).length) / rounds.length;
|
|
return {
|
|
median: q(0.5), p75: q(0.75), p90: q(0.9), p95: q(0.95), p99: q(0.99),
|
|
longest: rounds[rounds.length - 1],
|
|
over15: over(15), over20: over(20), over30: over(30),
|
|
wipeIfLong: (100 * longF.wiped) / Math.max(longF.n, 1),
|
|
wipeIfShort: (100 * shortF.wiped) / Math.max(shortF.n, 1),
|
|
longShare: (100 * longF.n) / Math.max(longF.n + shortF.n, 1),
|
|
capped, undecided
|
|
};
|
|
}
|
|
|
|
function measure(runs = RUNS) {
|
|
const out = {};
|
|
for (const cfg of CONFIGS) {
|
|
const party = roster(cfg.party);
|
|
const foes = Array(cfg.count).fill(creature(cfg.creature));
|
|
const each = SEEDS.map(s => sweep(party, foes, seedFor(s, cfg.id), runs));
|
|
const avg = f => r1(mean(each.map(f)));
|
|
out[cfg.id] = {
|
|
label: cfg.label, party: cfg.party, creature: cfg.creature, count: cfg.count,
|
|
median: avg(e => e.median), p75: avg(e => e.p75), p90: avg(e => e.p90),
|
|
p95: avg(e => e.p95), p99: avg(e => e.p99),
|
|
/* A single order statistic from one sample and the least stable number here: same
|
|
party, same seeds, same runs, and renaming this config from "line" to "column"
|
|
moved it 71 -> 90, because seedFor derives the stream from the id. Measured range
|
|
across seven labels: median 0, p95 1, p99 3, longest 21. Recorded for exact drift
|
|
only — check-cited refuses to let it be quoted, because a sample maximum reads
|
|
exactly like a limit. */
|
|
longest: Math.max(...each.map(e => e.longest)),
|
|
over15: avg(e => e.over15), over20: avg(e => e.over20), over30: avg(e => e.over30),
|
|
wipeIfLong: avg(e => e.wipeIfLong), wipeIfShort: avg(e => e.wipeIfShort),
|
|
longShare: avg(e => e.longShare),
|
|
/* Seed-to-seed spread of the deadlier ratio. Recorded because renaming this config
|
|
re-seeded it via seedFor and moved the six-a-side figure from 2.8 to 2.4 — three
|
|
seeds are not enough to quote this ratio to one decimal, only to assert its sign. */
|
|
deadlierNoise: r1(Math.max(...each.map(e => e.wipeIfLong / Math.max(e.wipeIfShort, 1e-9)))
|
|
- Math.min(...each.map(e => e.wipeIfLong / Math.max(e.wipeIfShort, 1e-9)))),
|
|
capped: each.reduce((a, e) => a + e.capped, 0),
|
|
undecided: each.reduce((a, e) => a + e.undecided, 0)
|
|
};
|
|
}
|
|
return out;
|
|
}
|
|
|
|
/** Refuse to record or trust a censored measurement. */
|
|
function assertUncensored(m, when) {
|
|
const bad = Object.entries(m).filter(([, c]) => c.capped > 0 || c.undecided > 0);
|
|
if (!bad.length) return;
|
|
console.error(`fight-tail: FAILED — ${when}: ${bad.map(([id, c]) =>
|
|
`${id} had ${c.capped} fights reach the ${CAP}-round ceiling and ${c.undecided} that ended `
|
|
+ `neither won nor wiped`).join("; ")}. The top of the distribution is truncated, so every `
|
|
+ `percentile above it is a wall rather than a measurement. Raise CAP.`);
|
|
process.exit(1);
|
|
}
|
|
|
|
if (UPDATE) {
|
|
const m = measure();
|
|
assertUncensored(m, "refusing to record");
|
|
await writeFile(BASELINE, JSON.stringify({
|
|
note: "Generated by tools/fight-tail.mjs --update. Do not edit by hand.",
|
|
runs: RUNS, seeds: SEEDS, cap: CAP, longRound: LONG, deadlierBar: DEADLIER,
|
|
configs: m
|
|
}, null, 2) + "\n", "utf8");
|
|
console.log(`fight-tail: baseline recorded — ${CONFIGS.length} encounters, ${RUNS} runs x ${SEEDS.length} seeds, nothing censored`);
|
|
process.exit(0);
|
|
}
|
|
|
|
if (!existsSync(BASELINE)) {
|
|
console.error("fight-tail: no baseline. Run with --update to record one.");
|
|
process.exit(1);
|
|
}
|
|
const base = JSON.parse(await readFile(BASELINE, "utf8"));
|
|
|
|
if (!CHECK) {
|
|
console.log(`\nhow long the fight runs — ${base.runs} runs x ${base.seeds.length} seeds, `
|
|
+ `nothing truncated (ceiling ${base.cap})\n`);
|
|
for (const c of Object.values(base.configs)) {
|
|
console.log(` ${c.label}`);
|
|
console.log(` median ${c.median} · 75th ${c.p75} · 90th ${c.p90} · 95th ${c.p95} · 99th ${c.p99}`);
|
|
console.log(` past 15 rounds ${c.over15}% · past 20 ${c.over20}% · past 30 ${c.over30}%`);
|
|
console.log(` the honest long-end figure is the 99th percentile, ${c.p99} — stable to about `
|
|
+ `three rounds across streams.`);
|
|
console.log(` (longest in this sample ${c.longest}, which is NOT a bound: the maximum of `
|
|
+ `6000 fights moves across a 21-round range on`);
|
|
console.log(` a re-seed and is recorded for drift only. Do not cite it.)`);
|
|
console.log(` a fight past 15 rounds wipes the party ${c.wipeIfLong}% of the time, `
|
|
+ `against ${c.wipeIfShort}% for shorter ones — ${r1(c.wipeIfLong / c.wipeIfShort)}x`);
|
|
console.log("");
|
|
}
|
|
console.log(` The six-a-side fight is the LONGER one (median ${base.configs.column.median} against `
|
|
+ `${base.configs.cut.median}). More bodies means more of them fighting on at reduced skill`);
|
|
console.log(` instead of dropping. The cut is shorter because it is decisive, not because it is safer.\n`);
|
|
process.exit(0);
|
|
}
|
|
|
|
/* ---- --check ------------------------------------------------------------------ */
|
|
|
|
if (base.runs !== RUNS || base.seeds?.join() !== SEEDS.join()
|
|
|| base.cap !== CAP || base.longRound !== LONG) {
|
|
console.error(`fight-tail: FAILED — the baseline was recorded under different conditions `
|
|
+ `(${base.runs} runs, seeds ${base.seeds?.join(",")}, ceiling ${base.cap}, long at ${base.longRound}). `
|
|
+ `Re-record it with --update.`);
|
|
process.exit(1);
|
|
}
|
|
for (const cfg of CONFIGS) {
|
|
const was = base.configs[cfg.id];
|
|
if (!was) { console.error(`fight-tail: FAILED — nothing recorded for "${cfg.id}". Re-record with --update.`); process.exit(1); }
|
|
if (was.party.join() !== cfg.party.join() || was.creature !== cfg.creature || was.count !== cfg.count) {
|
|
const gone = was.party.filter(k => !cfg.party.includes(k));
|
|
const added = cfg.party.filter(k => !was.party.includes(k));
|
|
console.error(`fight-tail: FAILED — ${cfg.id}: the scenario now casts different people`
|
|
+ (gone.length ? ` — no longer ${gone.join(", ")}` : "")
|
|
+ (added.length ? `, now ${added.join(", ")}` : "")
|
|
+ `. The recorded distribution describes the old cast; re-record with --update, `
|
|
+ `do not revert the cast.`);
|
|
process.exit(1);
|
|
}
|
|
}
|
|
|
|
/* An exact comparison is only honest if the measurement is exact. Prove it. */
|
|
{
|
|
const a = JSON.stringify(measure(150));
|
|
const b = JSON.stringify(measure(150));
|
|
if (a !== b) {
|
|
console.error("fight-tail: FAILED — the simulation is not deterministic, so an exact "
|
|
+ "baseline cannot mean anything.");
|
|
process.exit(1);
|
|
}
|
|
}
|
|
|
|
const now = measure();
|
|
assertUncensored(now, "the measurement no longer fits under the ceiling");
|
|
|
|
const problems = [];
|
|
for (const [id, was] of Object.entries(base.configs)) {
|
|
const m = now[id];
|
|
if (!m) { problems.push(` ${id}: no longer measured`); continue; }
|
|
for (const f of ["median", "p75", "p90", "p95", "p99", "longest",
|
|
"over15", "over20", "over30", "wipeIfLong", "wipeIfShort", "longShare", "deadlierNoise"]) {
|
|
if (m[f] !== was[f]) problems.push(` ${id}: ${f} ${was[f]} -> ${m[f]}`);
|
|
}
|
|
}
|
|
if (problems.length) {
|
|
console.error("fight-tail: FAILED — these fights no longer run the length they were recorded at");
|
|
problems.forEach(p => console.error(p));
|
|
console.error(" Read `node tools/fight-tail.mjs` before re-recording, and check whether "
|
|
+ "EXPOSURE's prose about the long end still holds.");
|
|
process.exit(1);
|
|
}
|
|
|
|
/* The two claims EXPOSURE is allowed to print, restated as tests. The drift comparison
|
|
above already catches every movement, so these assert only the sentences — and unlike
|
|
pinned figures they cannot be satisfied by re-recording, because --update rewrites the
|
|
numbers and leaves the shape. */
|
|
const notDeadlier = Object.entries(now).filter(([, c]) => c.wipeIfLong / c.wipeIfShort < DEADLIER);
|
|
if (notDeadlier.length) {
|
|
console.error(`fight-tail: FAILED — ${notDeadlier.map(([id]) => id).join(", ")}: a long fight is no `
|
|
+ `longer at least ${DEADLIER}x likelier to end in a wipe than a short one, so "a fight still `
|
|
+ `going is one the party is losing slowly" is false. Rewrite the prose rather than re-recording.`);
|
|
process.exit(1);
|
|
}
|
|
if (now.column.median <= now.cut.median) {
|
|
console.error(`fight-tail: FAILED — the six-a-side fight is no longer the longer one `
|
|
+ `(median ${now.column.median} against the cut's ${now.cut.median}). EXPOSURE says more bodies `
|
|
+ `means a longer fight, not a safer one; that sentence is now wrong. Rewrite it rather than `
|
|
+ `re-recording.`);
|
|
process.exit(1);
|
|
}
|
|
|
|
console.log(`fight-tail: OK — a fight past ${LONG} rounds wipes the party `
|
|
+ `${r1(now.cut.wipeIfLong / now.cut.wipeIfShort)}x as often in the cut and `
|
|
+ `${r1(now.column.wipeIfLong / now.column.wipeIfShort)}x at six a side, the six-a-side fight is the `
|
|
+ `longer one (median ${now.column.median} vs ${now.cut.median}), and nothing was truncated`);
|