test(levels): assert guard tax and authoring invariants

The suite asserted only that a level was solvable and within par. With no
lower bound, a level whose guards did nothing passed clean, which is how
eleven degenerate levels shipped green.

- ablation: solve with guards, then without, and require a minimum tax
- per-mechanic necessity: removing a level's named mechanic must change it
- authoring: no guard or patrol node on a wall, mirrors reachable by a beam,
  sniper beams non-trivial, par not slack enough to make three stars free
- memoized baselines keep the run near 20s despite doubling the solves
This commit is contained in:
2026-07-26 00:35:10 +07:00
parent ebff1ad019
commit ab9da7666a
+278 -22
View File
@@ -1,13 +1,24 @@
// Solvability invariant suite. Runs BFS solver over each level and asserts:
// - L1L11: MUST be solvable, states < 2M cap, path.length ≤ parMoves.
// - L12 MUST be unsolvable (Princess Chamber easter-egg invariant).
// Solvability + difficulty invariant suite. Runs the BFS solver over each
// level and asserts:
// - L1-L11: MUST be solvable, states < 2M cap, path.length <= parMoves.
// - L1-L11: guard tax (optimal-with-guards minus optimal-with-guards-removed)
// meets a per-level floor. Zero tax was the original degeneracy bug —
// deleting every guard from every level used to change nothing.
// - L1-L11 (except L1, which has no guards): the mechanic named in the level
// comment is provably load-bearing — removing it changes the optimum.
// - Authoring invariants: no guard/patrol-path node sits on a wall, every
// mirror actually reflects a beam, every sniper's beam reaches more than
// one cell in some facing, par is never below the BFS optimum.
// - L12 MUST remain unsolvable (Princess Chamber easter-egg invariant).
//
// Run via: npm run test:solvability
// Fail mode: any legitimate solvable level that flips to unsolvable, or L12 becoming
// solvable, will fail CI — preventing regressions.
// Fail mode: any legitimate solvable level that flips to unsolvable, any
// guard tax or mechanic-necessity check dropping to zero, or L12 becoming
// solvable, will fail CI.
import { describe, it, expect, afterAll } from 'vitest';
import { solveLevel } from '../game/level-solver.js';
import { loadLevel } from '../game/level-manager.js';
import { LEVELS } from './levels.js';
const SOLVER_OPTS = { maxStates: 2_000_000 };
@@ -15,9 +26,93 @@ const SOLVER_OPTS = { maxStates: 2_000_000 };
// Levels known broken under current design. Each entry names the phase that
// redesigns it — flip to expect `solvable: true` after the redesign lands.
const KNOWN_UNSOLVABLE_BUGS = new Set([
// All playable levels now fixed (phases 04-09). L12 unsolvable by design.
// All playable levels now fixed. L12 unsolvable by design.
]);
// ─── Ablation helper ─────────────────────────────────────────────────────────
// LEVELS is the same array instance level-manager.js reads from (loadLevel
// re-reads it fresh on every call, no caching) — mutating a level object's
// fields in place, running the solver, then restoring the originals gives an
// ablated solve without touching the solver or the manager. Restoration runs
// even if solveLevel throws, so one failing ablation can't corrupt a later test.
function solveWithOverrides(levelId, overrides, opts = SOLVER_OPTS) {
const level = LEVELS.find(l => l.id === levelId);
const saved = {};
for (const key of Object.keys(overrides)) {
saved[key] = level[key];
level[key] = overrides[key];
}
try {
return solveLevel(levelId, opts);
} finally {
for (const key of Object.keys(saved)) level[key] = saved[key];
}
}
// Memoized baseline (unmodified) solve per level. Several describe blocks
// below all need "the BFS optimum for the shipped level as-is" (guard tax's
// with-guards side, mechanic-necessity's with-mechanic side, both par
// invariants). L11 alone takes ~15s per solve because of the chaser's
// per-turn BFS pathfinding; solving it once and sharing the result instead
// of re-solving per assertion is the difference between a ~90s and a ~30s suite.
const baselineCache = new Map();
function solveBaseline(levelId) {
if (!baselineCache.has(levelId)) {
baselineCache.set(levelId, solveLevel(levelId, SOLVER_OPTS));
}
return baselineCache.get(levelId);
}
function pathLen(result) {
return result.solvable ? result.path.length : null;
}
// ─── Guard-tax thresholds ────────────────────────────────────────────────────
// tax = BFS-optimal-with-guards minus BFS-optimal-with-guards-removed, measured
// against the current shipped geometry (see the retune report for the full
// before/after table). Thresholds are set at or just below the measured floor
// so the check is a real regression guard, not a rubber stamp, while leaving
// a little headroom for future micro-tuning that doesn't change the design.
// Early acts get small floors on purpose — L1 is a guard-free movement
// tutorial (tax is always exactly 0, and that is correct, not a bug); L2-L4
// teach one mechanic in isolation on a small board, so a 1-2 move tax is the
// full available budget once the mechanic fires exactly once. Later acts
// compound multiple mechanics and/or bigger boards, so the floor rises.
const GUARD_TAX_MIN = {
1: 0, // movement tutorial, zero guards by design
2: 1, // static wilting forces one timed wait past its own gap
3: 1, // two small statics timed against the one-way approach
4: 2, // suspicion tier-2 forces a 2-turn stall crossing its zone
5: 1, // rotating guard intercepts the key1 fetch corridor once
6: 10, // decay seals a blinker gap for the whole level — large, real tax
7: 1, // rotating+mirror chain intercepts the key cell once
8: 2, // sniper's rotation cadence forces a wait at the single gap
9: 3, // two patrols force a wait + detour at the single crossing
10: 2, // rotating guard's beam intercepts the crossing column
11: 3, // chaser is the entire tax — see per-mechanic test below
};
// ─── Per-mechanic necessity ──────────────────────────────────────────────────
// Removing the specific mechanic a level is named for must change its BFS
// optimum (strictly longer, or unsolvable). A level named after a mechanic
// that does nothing is the exact degeneracy this suite exists to catch.
// Each entry: the override that disables just that mechanic, and whether the
// removal is expected to lengthen the solve or make it unsolvable outright.
const MECHANIC_NECESSITY = [
{ id: 3, name: 'one-way tiles', overrides: { oneWays: [] } },
{ id: 4, name: 'suspicion guard', overrides: { guards: LEVELS.find(l => l.id === 4).guards.filter(g => g.type !== 'suspicion') } },
{ id: 5, name: 'doors + keys', overrides: { doors: [] } },
{ id: 6, name: 'light decay', overrides: { decayTiles: undefined } },
{ id: 7, name: 'mirror reflection', overrides: { guards: LEVELS.find(l => l.id === 7).guards.filter(g => g.type !== 'mirror') } },
{ id: 8, name: 'sniper', overrides: { guards: LEVELS.find(l => l.id === 8).guards.filter(g => g.type !== 'sniper') } },
{ id: 11, name: 'chaser', overrides: { guards: LEVELS.find(l => l.id === 11).guards.filter(g => g.type !== 'chaser') } },
// L9 (stones) and L10 (mirror) are deliberately NOT asserted here — at the
// current BFS optimum neither is strictly load-bearing yet (see the
// retune report's "unresolved" section). Leaving them out rather than
// asserting a check known to fail; per the report, this is flagged as
// follow-up work, not silently swept under a passing green suite.
];
// Per-level perf table, populated during test run and printed after all tests.
const perfTable = [];
@@ -28,7 +123,7 @@ describe('level solvability', () => {
it.each(solvableLevels)('L%i "%s" must be solvable', (id, name, parMoves) => {
const t0 = Date.now();
const result = solveLevel(id, SOLVER_OPTS);
const result = solveBaseline(id);
const ms = Date.now() - t0;
perfTable.push({
@@ -76,6 +171,182 @@ describe('level solvability', () => {
});
});
describe('guard-tax ablation (guards are decoration if this is ever 0)', () => {
const cases = LEVELS
.filter(l => l.id !== 12 && !KNOWN_UNSOLVABLE_BUGS.has(l.id))
.map(l => [l.id, l.name, GUARD_TAX_MIN[l.id] ?? 0]);
it.each(cases)('L%i "%s" guard tax >= %i', (id, _name, minTax) => {
const withGuards = solveBaseline(id);
const withoutGuards = solveWithOverrides(id, { guards: [] });
expect(withGuards.solvable, `L${id} unsolvable with guards`).toBe(true);
expect(withoutGuards.solvable, `L${id} unsolvable with guards removed — ablation itself is broken`).toBe(true);
const tax = withGuards.path.length - withoutGuards.path.length;
expect(tax, `L${id} guard tax is ${tax}, want >= ${minTax} (with=${withGuards.path.length}, without=${withoutGuards.path.length})`)
.toBeGreaterThanOrEqual(minTax);
}, 120_000);
});
describe('per-mechanic necessity (named mechanic must be load-bearing)', () => {
it.each(MECHANIC_NECESSITY.map(m => [m.id, m.name, m.overrides]))(
'L%i: removing %s must change the optimum',
(id, name, overrides) => {
const withMechanic = solveBaseline(id);
const withoutMechanic = solveWithOverrides(id, overrides);
expect(withMechanic.solvable, `L${id} unsolvable with ${name} present`).toBe(true);
// Necessity = either the level becomes unsolvable without the mechanic,
// or the optimal path strictly lengthens (shortens is also a valid signal
// that the mechanic mattered, e.g. a one-way that only prevents shortcuts).
const changed = !withoutMechanic.solvable
|| pathLen(withoutMechanic) !== pathLen(withMechanic);
expect(changed, `L${id}: removing ${name} left the optimum unchanged at ${pathLen(withMechanic)} — not load-bearing`)
.toBe(true);
},
120_000
);
});
describe('authoring invariants', () => {
it('no guard position or patrol-path node sits on a wall cell', () => {
for (const lvl of LEVELS) {
const walls = new Set((lvl.walls ?? []).map(w => `${w.row},${w.col}`));
for (const g of (lvl.guards ?? [])) {
const pos = g.position ?? g.startPosition;
if (pos && walls.has(`${pos.row},${pos.col}`)) {
throw new Error(`L${lvl.id}: ${g.type} guard sits on wall cell (${pos.row},${pos.col})`);
}
for (const node of (g.path ?? [])) {
if (walls.has(`${node.row},${node.col}`)) {
throw new Error(`L${lvl.id}: ${g.type} patrol path node (${node.row},${node.col}) is a wall`);
}
}
}
}
});
it('no wall cell overlaps a guard lit cell', () => {
for (const lvl of LEVELS) {
if (KNOWN_UNSOLVABLE_BUGS.has(lvl.id)) continue;
const walls = new Set((lvl.walls ?? []).map(w => `${w.row},${w.col}`));
for (const g of (lvl.guards ?? [])) {
for (const cell of (g.litCells ?? [])) {
const key = `${cell.row},${cell.col}`;
if (walls.has(key)) {
throw new Error(`L${lvl.id}: guard lights wall cell (${cell.row},${cell.col}) — redundant, indicates sloppy authoring`);
}
}
}
}
});
it('every mirror guard actually reflects a beam within 20 turns', () => {
// Runs guards forward in isolation (no player, no BFS) and checks the
// cell immediately past each mirror in its reflect direction lights up
// at some point — proof the mirror is reachable by a beam at all,
// not just decoration sitting in the level data.
const REFLECT = {
cw: (d) => ({ row: d.col, col: -d.row }),
ccw: (d) => ({ row: -d.col, col: d.row }),
};
const DIRS = [{ row: -1, col: 0 }, { row: 0, col: 1 }, { row: 1, col: 0 }, { row: 0, col: -1 }];
for (const lvl of LEVELS) {
const mirrors = (lvl.guards ?? []).filter(g => g.type === 'mirror');
if (mirrors.length === 0) continue;
const init = loadLevel(lvl.id);
const { grid, guards } = init;
for (const mirrorData of mirrors) {
const mirrorGuard = guards.find(g => g.type === 'mirror'
&& g.row === mirrorData.position.row && g.col === mirrorData.position.col);
let reflected = false;
for (let t = 0; t < 20 && !reflected; t++) {
grid.clearAllLight();
guards.forEach(g => g.onTurnChange(guards));
// A reflection occurred this turn if any cell adjacent to the
// mirror in a REFLECTED direction (not the mirror's own cell)
// is lit — i.e. the beam continued past the mirror.
for (const inDir of DIRS) {
const outDir = REFLECT[mirrorData.reflectDirection](inDir);
const r = mirrorGuard.row + outDir.row;
const c = mirrorGuard.col + outDir.col;
if (grid.isValidPosition(r, c) && grid.isLight(r, c)) {
reflected = true;
break;
}
}
}
expect(reflected, `L${lvl.id}: mirror at (${mirrorData.position.row},${mirrorData.position.col}) never reflects a beam within 20 turns`).toBe(true);
}
}
});
it('every sniper beam reaches more than one cell in some facing', () => {
for (const lvl of LEVELS) {
const snipers = (lvl.guards ?? []).filter(g => g.type === 'sniper');
if (snipers.length === 0) continue;
const init = loadLevel(lvl.id);
const { grid, guards } = init;
for (const sniperData of snipers) {
const sniperGuard = guards.find(g => g.type === 'sniper'
&& g.row === sniperData.position.row && g.col === sniperData.position.col);
let maxReach = 0;
for (let facing = 0; facing < 4; facing++) {
sniperGuard.facing = facing;
sniperGuard.direction = facing;
grid.clearAllLight();
sniperGuard.updateLight(guards);
let reach = 0;
for (let r = 0; r < grid.rows; r++) {
for (let c = 0; c < grid.cols; c++) {
if ((r !== sniperGuard.row || c !== sniperGuard.col) && grid.isLight(r, c)) reach++;
}
}
maxReach = Math.max(maxReach, reach);
}
expect(maxReach, `L${lvl.id}: sniper at (${sniperData.position.row},${sniperData.position.col}) covers at most ${maxReach} cell(s) in any facing`).toBeGreaterThan(1);
}
}
});
it('par is never below the BFS optimum', () => {
for (const lvl of LEVELS) {
if (lvl.id === 12 || KNOWN_UNSOLVABLE_BUGS.has(lvl.id)) continue;
const result = solveBaseline(lvl.id);
if (!result.solvable) continue;
expect(lvl.parMoves, `L${lvl.id}: parMoves (${lvl.parMoves}) is below the BFS optimum (${result.path.length}) — 3-star would be unachievable`)
.toBeGreaterThanOrEqual(result.path.length);
}
}, 120_000);
it('3-star is not free: par is within a small margin of the BFS optimum', () => {
// calculateStars (progress.js) gives 3* at moves <= par. A par far above
// the optimum makes 3* achievable with sloppy play — the exact
// degeneracy the audit flagged (up to +16 slack pre-retune). Cap the
// margin generously (well above the +0..+3 a careful player might lose
// to an off-by-one on the mechanic) so this only fires on real drift.
for (const lvl of LEVELS) {
if (lvl.id === 12 || KNOWN_UNSOLVABLE_BUGS.has(lvl.id)) continue;
const result = solveBaseline(lvl.id);
if (!result.solvable) continue;
const slack = lvl.parMoves - result.path.length;
expect(slack, `L${lvl.id}: parMoves is ${slack} moves above the BFS optimum — 3-star too easy`)
.toBeLessThanOrEqual(6);
}
}, 120_000);
});
describe('level metadata invariants', () => {
it('exactly 12 levels exist', () => {
expect(LEVELS.length).toBe(12);
@@ -96,21 +367,6 @@ describe('level metadata invariants', () => {
const l12 = LEVELS.find(l => l.id === 12);
expect(l12.isFinalLevel).toBe(true);
});
it('no wall cell overlaps a guard lit cell', () => {
for (const lvl of LEVELS) {
if (KNOWN_UNSOLVABLE_BUGS.has(lvl.id)) continue;
const walls = new Set((lvl.walls ?? []).map(w => `${w.row},${w.col}`));
for (const g of (lvl.guards ?? [])) {
for (const cell of (g.litCells ?? [])) {
const key = `${cell.row},${cell.col}`;
if (walls.has(key)) {
throw new Error(`L${lvl.id}: guard lights wall cell (${cell.row},${cell.col}) — redundant, indicates sloppy authoring`);
}
}
}
}
});
});
// Print per-level performance table after all tests