voyage/tests/lib/research-loop-cap.test.mjs
Kjell Tore Guttormsen 7dc0add768 feat(research-loop-cap): give the discovery ceiling a reader, not just a sentence
The bounded-cost NFR asks for explicit ceilings on BOTH axes - max
conversation turns and max discovered dimensions. The turn axis got
MAX_CONV_TURNS, a ledger-backed reader and a PreToolUse enforcer. The
discovery axis got one sentence in Phase 4.5 prose ("append candidates only
while the whole list stays at or below maxDimensions: 8") with no constant
of its own, no reader, and no test that a run exceeding it is caught. That
is the brief_reviewer_iter_cap shape the operator decision warned about: a
cap nothing reads.

checkDimensionCeiling() is the reader, exposed on the CLI as
--check-dimensions N (exit 0 within, exit 1 rejected), and Phase 4.5 step 3
now calls it once the final list is settled instead of merely describing the
bound.

Three deliberate choices:

- The ceiling IS MAX_TOTAL_DIMENSIONS, the constant that sizes the turn
  budget. Both axes read one settings.json:16 value, so they cannot end up
  enforcing different numbers - a second constant is how that drift starts.
- An unreadable count is REJECTED ('abc', null, undefined, {}, -1, NaN,
  non-integers). A cost ceiling that waves through what it cannot measure is
  not a ceiling.
- --check-dimensions requires no run id, effort or VOYAGE_STORM_ENABLED.
  Phase 4.5 never calls the budget gate - that is why its skip-guard reads
  the flag directly - so the ceiling check must not inherit the gate's
  preconditions.

The mitigation the review already verified still holds and is unchanged:
MAX_TOTAL_DIMENSIONS bounds actual retrieval cost regardless of how many
dimensions discovery appends. What was missing was anything that FAILS on a
list over the bound, and now a run over it is rejected by exit code.

Review finding 96a3ee51152dfe72aca703f771843f2f3639e7b6 (MINOR).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LuGhWAbWyRFBFeemfhxoVv
2026-08-12 23:09:11 +02:00

597 lines
26 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

// tests/lib/research-loop-cap.test.mjs
// Cover lib/util/research-loop-cap.mjs: default-off, worst-case arithmetic,
// anti-dead-data (different caps → different denial points), statefulness
// (identical args → different answers once the budget is hit), env
// coercion, fail-closed on missing CLAUDE_PLUGIN_DATA, and the CLI shim.
import { test } from 'node:test';
import { strict as assert } from 'node:assert';
import { execFileSync, execFile } from 'node:child_process';
import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, rmSync, existsSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { dirname, join } from 'node:path';
import { fileURLToPath } from 'node:url';
import {
allowTurn,
isStormEnabled,
resolveMaxConvTurns,
resolveLedgerPath,
resolveDataRoot,
readLedger,
checkDimensionCeiling,
MAX_CONV_TURNS,
MAX_TOTAL_DIMENSIONS,
} from '../../lib/util/research-loop-cap.mjs';
const HERE = dirname(fileURLToPath(import.meta.url));
const SHIM = join(HERE, '..', '..', 'lib', 'util', 'research-loop-cap.mjs');
function withTmpDataDir(fn) {
const dir = mkdtempSync(join(tmpdir(), 'research-loop-cap-'));
try {
return fn(dir);
} finally {
rmSync(dir, { recursive: true, force: true });
}
}
function runShim(args, env) {
try {
const out = execFileSync(process.execPath, [SHIM, ...args], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'pipe'],
env: { ...process.env, ...env },
});
return { code: 0, out };
} catch (e) {
return { code: e.status ?? 1, out: e.stdout?.toString() ?? '' };
}
}
/**
* Run the shim in a BUILT env rather than an inherited one. Spreading
* process.env means no test can express "CLAUDE_PLUGIN_DATA is absent" — the
* exact condition that holds in every real run — so the shim's behaviour there
* went uncovered while the module was denying turn 1.
*/
function runShimStripped(args, env = {}) {
try {
const out = execFileSync(process.execPath, [SHIM, ...args], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'pipe'],
env: { PATH: process.env.PATH, ...env },
});
return { code: 0, out };
} catch (e) {
return { code: e.status ?? 1, out: e.stdout?.toString() ?? '' };
}
}
// ---- (a) default-off --------------------------------------------------------
test('allowTurn — VOYAGE_STORM_ENABLED unset denies with budget 0, regardless of effort', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir };
const r = allowTurn({ runId: 'r1', dimension: 'd1', effort: 'high' }, { env });
assert.equal(r.ok, false);
assert.equal(r.reason, 'storm_disabled');
assert.equal(r.budget, 0);
});
});
test('allowTurn — VOYAGE_STORM_ENABLED=0 denies same as unset', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '0' };
const r = allowTurn({ runId: 'r1', dimension: 'd1', effort: 'high' }, { env });
assert.equal(r.ok, false);
assert.equal(r.reason, 'storm_disabled');
});
});
test('allowTurn — enabled but effort !== high denies with budget 0', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1' };
const r = allowTurn({ runId: 'r1', dimension: 'd1', effort: 'standard' }, { env });
assert.equal(r.ok, false);
assert.equal(r.reason, 'effort_not_high');
assert.equal(r.budget, 0);
});
});
// ---- (f) CLAUDE_PLUGIN_DATA unset => documented fallback root ----------------
//
// CLAUDE_PLUGIN_DATA is EMPTY in the Bash tool's process env (measured in a
// live plugin-enabled session), and the Bash snippet in commands/trekresearch.md
// is the module's only caller. Denying on its absence therefore denied turn 1
// of every real run: the loop could never spend a turn, and the pre-registered
// measurement could not be run at all. The root is resolved in code, not
// demanded of the environment.
test('allowTurn — CLAUDE_PLUGIN_DATA unset falls back to the documented root and grants', () => {
withTmpDataDir((home) => {
const env = { VOYAGE_STORM_ENABLED: '1', HOME: home }; // no CLAUDE_PLUGIN_DATA
const r = allowTurn({ runId: 'r1', dimension: 'd1', effort: 'high' }, { env });
assert.equal(r.ok, true, 'the loop must be able to spend turn 1 without CLAUDE_PLUGIN_DATA');
assert.equal(r.used, 1);
assert.equal(r.budget, MAX_CONV_TURNS * MAX_TOTAL_DIMENSIONS);
assert.ok(
existsSync(join(home, '.claude', 'voyage', 'trekresearch-loop-ledger.jsonl')),
'the ledger must be written under the fallback root',
);
});
});
test('resolveDataRoot — CLAUDE_PLUGIN_DATA wins; empty or unset falls back to ~/.claude/voyage', () => {
assert.equal(resolveDataRoot({ CLAUDE_PLUGIN_DATA: '/tmp/plugin-data' }), '/tmp/plugin-data');
assert.equal(resolveDataRoot({ CLAUDE_PLUGIN_DATA: '', HOME: '/home/x' }), join('/home/x', '.claude', 'voyage'));
assert.equal(resolveDataRoot({ HOME: '/home/x' }), join('/home/x', '.claude', 'voyage'));
});
// ---- (c) worst-case arithmetic ----------------------------------------------
test('allowTurn — budget is max_conv_turns × max_total_dimensions (default 3×8=24)', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1' };
const r = allowTurn({ runId: 'r1', dimension: 'd1', effort: 'high' }, { env });
assert.equal(r.ok, true);
assert.equal(r.budget, 24);
assert.equal(r.used, 1);
});
});
test('allowTurn — grants exactly `budget` turns then denies the next one (default 24)', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1' };
let last;
for (let i = 0; i < 24; i++) {
last = allowTurn({ runId: 'r-exhaust', dimension: `d${i % 8}`, effort: 'high' }, { env });
assert.equal(last.ok, true, `turn ${i + 1} should be granted`);
}
const denied = allowTurn({ runId: 'r-exhaust', dimension: 'd0', effort: 'high' }, { env });
assert.equal(denied.ok, false);
assert.equal(denied.reason, 'budget_exhausted');
assert.equal(denied.used, 24);
assert.equal(denied.budget, 24);
});
});
// ---- (c)/(anti-dead-data) — different caps → observably different denial points
test('allowTurn — TREKRESEARCH_MAX_CONV_TURNS=1 denies after 8 turns (1×8), not 24', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
let last;
for (let i = 0; i < 8; i++) {
last = allowTurn({ runId: 'r-narrow', dimension: `d${i}`, effort: 'high' }, { env });
assert.equal(last.ok, true, `turn ${i + 1} should be granted`);
}
const denied = allowTurn({ runId: 'r-narrow', dimension: 'd8', effort: 'high' }, { env });
assert.equal(denied.ok, false);
assert.equal(denied.budget, 8);
assert.notEqual(denied.budget, 24, 'a narrower cap must produce a different denial point than the default');
});
});
// ---- (d) stateful — identical args give different answers once exhausted ---
test('allowTurn — identical {runId, dimension, effort} args diverge once the budget is hit (proves statefulness)', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
const args = { runId: 'r-identical', dimension: 'same-dim', effort: 'high' };
const results = [];
for (let i = 0; i < 9; i++) results.push(allowTurn(args, { env }));
// First 8 (budget = 1*8) granted, 9th denied — same exact input object each time.
assert.deepEqual(results.slice(0, 8).map(r => r.ok), Array(8).fill(true));
assert.equal(results[8].ok, false);
assert.equal(results[8].reason, 'budget_exhausted');
});
});
// ---- (e) env coercion --------------------------------------------------------
test('resolveMaxConvTurns — NaN string falls back to default', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: 'abc' }), MAX_CONV_TURNS);
});
test('resolveMaxConvTurns — empty string falls back to default', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '' }), MAX_CONV_TURNS);
});
test('resolveMaxConvTurns — negative value falls back to default', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '-5' }), MAX_CONV_TURNS);
});
test('resolveMaxConvTurns — zero falls back to default (never unbounded)', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '0' }), MAX_CONV_TURNS);
});
test('resolveMaxConvTurns — unset falls back to default', () => {
assert.equal(resolveMaxConvTurns({}), MAX_CONV_TURNS);
});
test('resolveMaxConvTurns — valid positive integer string is honored', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '2' }), 2);
});
// A fractional value passed the `n <= 0` guard and only THEN floored, so 0.5 and
// 0.9 became 0 and the budget became 0 × 8 = 0 — every turn denied, the loop
// silently dead, while README.md and docs/architecture.md both promise a
// fallback of 3. The guard has to see the floored value, not the raw one.
test('resolveMaxConvTurns — a fractional value below 1 falls back to the default, never 0', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '0.5' }), MAX_CONV_TURNS);
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '0.9' }), MAX_CONV_TURNS);
});
test('resolveMaxConvTurns — a fractional value above 1 still floors (2.7 → 2)', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: '2.7' }), 2);
});
test('resolveMaxConvTurns — Infinity is not a cap and falls back to the default', () => {
assert.equal(resolveMaxConvTurns({ TREKRESEARCH_MAX_CONV_TURNS: 'Infinity' }), MAX_CONV_TURNS);
});
test('allowTurn — a fractional cap below 1 cannot produce a budget of 0', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '0.5' };
const r = allowTurn({ runId: 'r-frac', dimension: 'd1', effort: 'high' }, { env });
assert.equal(r.ok, true, 'a budget of 0 would make the loop silently dead, not bounded');
assert.equal(r.budget, MAX_CONV_TURNS * MAX_TOTAL_DIMENSIONS);
});
});
// ---- pure-core unit coverage --------------------------------------------------
test('isStormEnabled — only the literal string "1" enables', () => {
assert.equal(isStormEnabled({ VOYAGE_STORM_ENABLED: '1' }), true);
assert.equal(isStormEnabled({ VOYAGE_STORM_ENABLED: 'true' }), false);
assert.equal(isStormEnabled({}), false);
});
test('resolveLedgerPath — falls back under ~/.claude/voyage when CLAUDE_PLUGIN_DATA is unset or empty', () => {
const expected = join('/home/x', '.claude', 'voyage', 'trekresearch-loop-ledger.jsonl');
assert.equal(resolveLedgerPath({ HOME: '/home/x' }), expected);
assert.equal(resolveLedgerPath({ CLAUDE_PLUGIN_DATA: '', HOME: '/home/x' }), expected);
});
test('resolveLedgerPath — joins CLAUDE_PLUGIN_DATA with the ledger filename', () => {
const p = resolveLedgerPath({ CLAUDE_PLUGIN_DATA: '/tmp/plugin-data' });
assert.equal(p, join('/tmp/plugin-data', 'trekresearch-loop-ledger.jsonl'));
});
test('allowTurn — missing runId or dimension denies with missing_args', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1' };
const r1 = allowTurn({ dimension: 'd1', effort: 'high' }, { env });
assert.equal(r1.ok, false);
assert.equal(r1.reason, 'missing_args');
const r2 = allowTurn({ runId: 'r1', effort: 'high' }, { env });
assert.equal(r2.ok, false);
assert.equal(r2.reason, 'missing_args');
});
});
// ---- fail-closed on an unreadable ledger ------------------------------------
//
// The module's own header states that a budget control "must never silently
// grant unlimited turns just because the data dir is missing". The
// missing-DIRECTORY case already failed closed; the unreadable-FILE case
// returned 0 from countTurns and therefore re-granted the full budget on every
// call, unbounded — a fail-open in the same module that argues against one.
// A directory standing where the ledger file belongs reproduces it portably
// (EISDIR), with no chmod that a root test runner would ignore.
function withUnreadableLedger(fn) {
return withTmpDataDir((dir) => {
mkdirSync(join(dir, 'trekresearch-loop-ledger.jsonl'), { recursive: true });
return fn(dir);
});
}
test('readLedger — a missing ledger is 0 turns, not an error (turn 1 must be grantable)', () => {
withTmpDataDir((dir) => {
assert.equal(readLedger(join(dir, 'nope.jsonl'), 'r1').granted, 0);
});
});
test('readLedger — an unreadable ledger throws rather than reporting 0 turns spent', () => {
withUnreadableLedger((dir) => {
assert.throws(
() => readLedger(join(dir, 'trekresearch-loop-ledger.jsonl'), 'r1'),
/unreadable/i,
);
});
});
test('readLedger — malformed lines are skipped, well-formed ones for the run still count', () => {
withTmpDataDir((dir) => {
const p = join(dir, 'trekresearch-loop-ledger.jsonl');
writeFileSync(p, [
JSON.stringify({ runId: 'r1', dimension: 'd1' }),
'{ not json',
'',
JSON.stringify({ runId: 'other', dimension: 'd1' }),
JSON.stringify({ runId: 'r1', dimension: 'd2' }),
].join('\n') + '\n');
assert.equal(readLedger(p, 'r1').granted, 2);
});
});
test('allowTurn — an unreadable ledger DENIES the turn instead of granting a fresh budget', () => {
withUnreadableLedger((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1' };
const r = allowTurn({ runId: 'r-unreadable', dimension: 'd1', effort: 'high' }, { env });
assert.equal(r.ok, false, 'a budget control that cannot count must not grant');
assert.match(r.reason, /ledger-read-failed/);
});
});
// ---- the bound holds under concurrency --------------------------------------
//
// countTurns-then-appendFileSync had no atomic claim, while the comment above
// allowTurn asserted "Append-only: never read-modify-write" and named the
// concurrent case (Phase 4.5/5 may spawn several agents in one message) as the
// reason. The decision path WAS read-then-write: N callers that all observe
// used == budget-1 all grant, and the bound is exceeded by N-1.
//
// Two of these tests are deterministic. They do not race anything — they assert
// the invariant the claim introduces: a slot that is already claimed is spent,
// even when the ledger has not caught up yet, which is exactly the state a
// mid-append competitor leaves behind. The third runs real processes.
function seedLedger(dir, runId, n) {
writeFileSync(
join(dir, 'trekresearch-loop-ledger.jsonl'),
Array.from({ length: n }, (_, i) =>
JSON.stringify({ ts: new Date().toISOString(), runId, dimension: `d${i}`, effort: 'high' }),
).join('\n') + (n ? '\n' : ''),
);
}
function seedClaims(dir, runId, slots) {
const claimDir = join(dir, 'trekresearch-loop-claims');
mkdirSync(claimDir, { recursive: true });
for (const s of slots) writeFileSync(join(claimDir, `${runId}-${s}.claim`), '');
}
function runShimAsync(args, env) {
return new Promise((resolve) => {
execFile(
process.execPath,
[SHIM, ...args],
{ encoding: 'utf-8', env: { PATH: process.env.PATH, ...env } },
(err, stdout) => resolve({ code: err ? (err.code ?? 1) : 0, out: stdout ?? '' }),
);
});
}
test('allowTurn — a slot already claimed is spent even when the ledger has not caught up', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
// budget = 1 × 8 = 8. Ledger shows 7 turns; a competitor already claimed
// slot 8 and has not appended yet. Counting the ledger alone says "one slot
// free" and grants a 9th turn overall — the breach this closes.
seedLedger(dir, 'r-race', 7);
seedClaims(dir, 'r-race', [1, 2, 3, 4, 5, 6, 7, 8]);
const r = allowTurn({ runId: 'r-race', dimension: 'd0', effort: 'high' }, { env });
assert.equal(r.ok, false, 'every slot up to the budget is claimed, so there is nothing to grant');
assert.equal(r.reason, 'budget_exhausted');
assert.equal(r.budget, 8);
});
});
test('allowTurn — it takes the first FREE slot and claims it, so a repeat call cannot retake it', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
seedLedger(dir, 'r-slot', 7);
seedClaims(dir, 'r-slot', [1, 2, 3, 4, 5, 6, 7]);
const first = allowTurn({ runId: 'r-slot', dimension: 'd0', effort: 'high' }, { env });
assert.equal(first.ok, true, 'slot 8 is free and must be grantable');
assert.equal(first.used, 8, 'used is the slot number, so it never double-counts a claimed slot');
assert.ok(
existsSync(join(dir, 'trekresearch-loop-claims', 'r-slot-8.claim')),
'the grant must leave the claim behind as the atomic record of the slot',
);
const second = allowTurn({ runId: 'r-slot', dimension: 'd0', effort: 'high' }, { env });
assert.equal(second.ok, false);
assert.equal(second.reason, 'budget_exhausted');
});
});
test('allowTurn — a different runId is unaffected by another runs claims', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
seedClaims(dir, 'r-other', [1, 2, 3, 4, 5, 6, 7, 8]);
const r = allowTurn({ runId: 'r-mine', dimension: 'd0', effort: 'high' }, { env });
assert.equal(r.ok, true, 'claims are per-run; one run must not exhaust another');
});
});
test('allowTurn — parallel processes at the boundary cannot exceed the budget', async () => {
const dir = mkdtempSync(join(tmpdir(), 'research-loop-cap-par-'));
try {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
seedLedger(dir, 'r-par', 7); // budget 8 → exactly one turn left
const results = await Promise.all(
Array.from({ length: 6 }, (_, i) =>
runShimAsync(['--run-id', 'r-par', '--dimension', `p${i}`, '--effort', 'high'], env),
),
);
const winners = results.filter((r) => r.code === 0).length;
assert.equal(winners, 1, `exactly one of six concurrent callers may take the last slot, got ${winners}`);
// Granted turns, not raw lines: the five denied callers also record
// exhaustion tombstones, and a tombstone is not a turn.
const { granted } = readLedger(join(dir, 'trekresearch-loop-ledger.jsonl'), 'r-par');
assert.equal(granted, 8, `granted turns must never exceed the budget, got ${granted}`);
} finally {
rmSync(dir, { recursive: true, force: true });
}
});
// ---- the exhaustion tombstone -----------------------------------------------
//
// allowTurn appends BEFORE the turn runs, so during granted turn N the ledger
// holds N records. The hook denied at `used >= budget`, which blocked every
// tool call of the FINAL granted turn: the primitive granted B turns and the
// harness permitted B-1. Worse, an exhausted run then always terminated through
// an exit-2 tool denial instead of the graceful "cap exhausted" exit at
// commands/trekresearch.md, which is the only exit the prose teaches.
//
// Letting the hook allow at `used == budget` fixes the count but would leave it
// unable to catch the one case it exists for — the loop consults the gate, is
// denied, and issues the tool call anyway. So the denial itself becomes a
// record: a tombstone the hook can see.
test('allowTurn — denying for budget writes an exhaustion tombstone', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
for (let i = 0; i < 8; i++) {
assert.equal(allowTurn({ runId: 'r-tomb', dimension: `d${i}`, effort: 'high' }, { env }).ok, true);
}
const ledgerPath = join(dir, 'trekresearch-loop-ledger.jsonl');
assert.equal(readLedger(ledgerPath, 'r-tomb').exhausted, 0, 'no tombstone before the gate has denied anything');
const denied = allowTurn({ runId: 'r-tomb', dimension: 'd0', effort: 'high' }, { env });
assert.equal(denied.ok, false);
assert.equal(denied.reason, 'budget_exhausted');
const after = readLedger(ledgerPath, 'r-tomb');
assert.equal(after.exhausted, 1, 'the denial must leave a record the harness gate can read');
assert.equal(after.granted, 8, 'a tombstone is not a granted turn and must not count as one');
});
});
test('allowTurn — the tombstone is written once, not once per repeated denial', () => {
withTmpDataDir((dir) => {
const env = { CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
for (let i = 0; i < 8; i++) allowTurn({ runId: 'r-once', dimension: `d${i}`, effort: 'high' }, { env });
for (let i = 0; i < 5; i++) allowTurn({ runId: 'r-once', dimension: 'd0', effort: 'high' }, { env });
const after = readLedger(join(dir, 'trekresearch-loop-ledger.jsonl'), 'r-once');
assert.equal(after.exhausted, 1, 'a hammered gate must not grow the ledger without bound');
assert.equal(after.granted, 8);
});
});
test('readLedger — a tombstone is reported separately and never as a granted turn', () => {
withTmpDataDir((dir) => {
const p = join(dir, 'trekresearch-loop-ledger.jsonl');
writeFileSync(p, [
JSON.stringify({ runId: 'r1', dimension: 'd1', slot: 1 }),
JSON.stringify({ runId: 'r1', exhausted: true }),
JSON.stringify({ runId: 'other', exhausted: true }),
].join('\n') + '\n');
const l = readLedger(p, 'r1');
assert.equal(l.granted, 1);
assert.equal(l.exhausted, 1);
assert.equal(readLedger(p, 'other').granted, 0, 'another runs tombstone is not a granted turn either');
});
});
// ---- the discovery ceiling has a reader, not just a sentence ----------------
//
// The bounded-cost NFR asks for explicit ceilings on BOTH axes: max conversation
// turns and max discovered dimensions. The turn axis got MAX_CONV_TURNS, a
// ledger-backed reader and a PreToolUse enforcer. The discovery axis got a
// sentence in Phase 4.5 — "append candidates only while the whole list stays at
// or below maxDimensions: 8" — with no constant of its own, no reader, and no
// test that a run exceeding it is caught. That is the same shape as the
// brief_reviewer_iter_cap failure the operator decision warned about: a cap
// nothing reads.
//
// The ceiling is deliberately the SAME constant that sizes the turn budget. Two
// constants for one settings.json:16 value is how the two drift apart.
test('checkDimensionCeiling — a list at the ceiling is accepted', () => {
const r = checkDimensionCeiling(Array.from({ length: MAX_TOTAL_DIMENSIONS }, (_, i) => `d${i}`));
assert.equal(r.ok, true);
assert.equal(r.count, MAX_TOTAL_DIMENSIONS);
assert.equal(r.ceiling, MAX_TOTAL_DIMENSIONS);
});
test('checkDimensionCeiling — one dimension over the ceiling is REJECTED', () => {
const r = checkDimensionCeiling(Array.from({ length: MAX_TOTAL_DIMENSIONS + 1 }, (_, i) => `d${i}`));
assert.equal(r.ok, false, 'a ceiling that accepts ceiling+1 is not a ceiling');
assert.equal(r.reason, 'ceiling_exceeded');
assert.equal(r.count, MAX_TOTAL_DIMENSIONS + 1);
});
test('checkDimensionCeiling — a plain count works as well as a list', () => {
assert.equal(checkDimensionCeiling(8).ok, true);
assert.equal(checkDimensionCeiling(9).ok, false);
assert.equal(checkDimensionCeiling('8').ok, true);
});
test('checkDimensionCeiling — an unreadable count is rejected, never waved through', () => {
for (const bad of ['abc', null, undefined, {}, -1, NaN]) {
const r = checkDimensionCeiling(bad);
assert.equal(r.ok, false, `${JSON.stringify(bad)} must not pass a cost ceiling`);
assert.equal(r.reason, 'unreadable_dimension_count');
}
});
test('checkDimensionCeiling — the ceiling is the same constant that sizes the turn budget', () => {
// Phase 4.5 and the Phase 5 budget must not be able to disagree about 8.
assert.equal(checkDimensionCeiling(0).ceiling, MAX_TOTAL_DIMENSIONS);
});
test('CLI shim — --check-dimensions exits 0 at the ceiling and 1 above it', () => {
const at = runShim(['--check-dimensions', String(MAX_TOTAL_DIMENSIONS)], {});
assert.equal(at.code, 0, `at the ceiling must exit 0; got ${at.out}`);
assert.equal(JSON.parse(at.out.trim()).ok, true);
const over = runShim(['--check-dimensions', String(MAX_TOTAL_DIMENSIONS + 1)], {});
assert.equal(over.code, 1, 'a run over the ceiling must be rejected by exit code, not by prose');
const parsed = JSON.parse(over.out.trim());
assert.equal(parsed.ok, false);
assert.equal(parsed.reason, 'ceiling_exceeded');
});
test('CLI shim — --check-dimensions needs no runId, effort or STORM flag', () => {
// It is a cost ceiling on Phase 4.5, which never calls the budget gate, so it
// must not inherit the budget gate's preconditions.
const r = runShimStripped(['--check-dimensions', '3'], {});
assert.equal(r.code, 0, `should not require --run-id/--effort; got ${r.out}`);
});
// ---- (g) shim contract --------------------------------------------------------
test('CLI shim — grants and exits 0 when enabled + high effort + budget available', () => {
withTmpDataDir((dir) => {
const r = runShim(
['--run-id', 'shim-1', '--dimension', 'd1', '--effort', 'high'],
{ CLAUDE_PLUGIN_DATA: dir, VOYAGE_STORM_ENABLED: '1' },
);
assert.equal(r.code, 0);
const parsed = JSON.parse(r.out.trim());
assert.equal(parsed.ok, true);
});
});
test('CLI shim — denies and exits 1 when disabled', () => {
const r = runShim(['--run-id', 'shim-2', '--dimension', 'd1', '--effort', 'high'], { VOYAGE_STORM_ENABLED: '0' });
assert.equal(r.code, 1);
const parsed = JSON.parse(r.out.trim());
assert.equal(parsed.ok, false);
assert.equal(parsed.reason, 'storm_disabled');
});
test('CLI shim — grants with CLAUDE_PLUGIN_DATA STRIPPED from the environment', () => {
withTmpDataDir((home) => {
const r = runShimStripped(
['--run-id', 'shim-stripped', '--dimension', 'd1', '--effort', 'high'],
{ VOYAGE_STORM_ENABLED: '1', HOME: home },
);
assert.equal(r.code, 0, `shim must grant without CLAUDE_PLUGIN_DATA; got: ${r.out}`);
const parsed = JSON.parse(r.out.trim());
assert.equal(parsed.ok, true);
assert.ok(existsSync(join(home, '.claude', 'voyage', 'trekresearch-loop-ledger.jsonl')));
});
});
test('CLI shim — missing required args exits 1 with usage reason', () => {
const r = runShim(['--run-id', 'shim-3']);
assert.equal(r.code, 1);
const parsed = JSON.parse(r.out.trim());
assert.equal(parsed.ok, false);
assert.match(parsed.reason, /usage:/);
});