allowTurn() appends BEFORE the turn runs, so during granted turn N the
ledger holds N records. The hook denied at `used >= budget`, which blocked
every tool call of the FINAL granted turn: the primitive granted B turns
and the harness permitted B-1. Worse, an exhausted run therefore always
terminated through an exit-2 tool denial instead of the graceful "cap
exhausted" exit at commands/trekresearch.md - and the prose says in as
many words that exit 2 is not exit 1, so the model was pushed out through
the one exit it is told NOT to treat as a cap.
The review recommended denying at `used > budget`. Taken alone that fixes
the count and breaks the hook: once the O_EXCL claim (previous commit)
makes a breached ledger impossible, `granted > budget` can no longer fire,
and the case this hook exists for - the loop consults the gate, is denied,
and issues the tool call anyway - would be allowed. A deny branch that
cannot be reached is a dead security claim, which is the same thing S82
removed two of rather than leave standing.
So the denial itself became a record. allowTurn() appends a tombstone
{runId, exhausted: true} when it denies for budget, and the hook denies on
the tombstone. Both properties now hold at once:
granted == budget, no tombstone -> turn B is in flight -> ALLOW
tombstone present -> the gate already said no -> DENY
granted > budget -> breached, any cause -> DENY
A tombstone is not a turn: readLedger reports {granted, exhausted}
separately so it can never consume budget. allowTurn short-circuits on an
existing tombstone, so a hammered gate neither re-walks every slot nor
grows the ledger. The tombstone write is best effort on purpose - the
denial is already the correct answer, so a ledger that cannot take the
record must not turn a denial into a grant.
The parallel-boundary test now asserts GRANTED turns rather than raw
ledger lines, because the denied callers legitimately add tombstones.
Review finding 8eb53458ac3efec778094f9f03b09e1cc1077a09 (MINOR).
Operator decision: tombstone over the literal recommended_action.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LuGhWAbWyRFBFeemfhxoVv
321 lines
14 KiB
JavaScript
321 lines
14 KiB
JavaScript
// tests/hooks/agent-cap.test.mjs
|
||
// Step 10 — pins hooks/scripts/pre-agent-cap.mjs, the PreToolUse enforcement
|
||
// of the /trekresearch Phase 5 loop bound.
|
||
//
|
||
// The spike (docs/spike-pretooluse-subagent-reach.md, RESULT: FIRES) proved a
|
||
// plugin PreToolUse hook observes sub-agent tool calls, so the cap can be
|
||
// enforced rather than merely documented. This file pins the two properties
|
||
// that matter in opposite directions:
|
||
//
|
||
// (a) it DENIES (exit 2) once the ledger shows the budget spent, and
|
||
// (b) it does NOT over-block — an unrelated session, unparsable stdin, a
|
||
// stale marker, or the kill switch all exit 0.
|
||
//
|
||
// Pattern: tests/hooks/bash-guard.test.mjs (child process via runHook).
|
||
|
||
import { test } from 'node:test';
|
||
import { strict as assert } from 'node:assert';
|
||
import { dirname, join } from 'node:path';
|
||
import { fileURLToPath } from 'node:url';
|
||
import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, rmSync } from 'node:fs';
|
||
import { tmpdir } from 'node:os';
|
||
import { runHookWithEnv } from '../helpers/hook-helper.mjs';
|
||
|
||
const HERE = dirname(fileURLToPath(import.meta.url));
|
||
const ROOT = join(HERE, '..', '..');
|
||
const CAP_HOOK = join(ROOT, 'hooks', 'scripts', 'pre-agent-cap.mjs');
|
||
const HOOKS_JSON = join(ROOT, 'hooks', 'hooks.json');
|
||
|
||
const SESSION = 'sess-abc123';
|
||
const RUN_ID = 'run-xyz789';
|
||
|
||
/**
|
||
* Build a throwaway CLAUDE_PLUGIN_DATA dir holding a scope marker for
|
||
* `sessionId` and `turns` ledger entries for RUN_ID.
|
||
*/
|
||
function fixture({ turns = 0, sessionId = SESSION, startedAt = new Date(), exhausted = false } = {}) {
|
||
const dir = mkdtempSync(join(tmpdir(), 'voyage-cap-'));
|
||
mkdirSync(join(dir, 'trekresearch-loop-scope'), { recursive: true });
|
||
writeFileSync(
|
||
join(dir, 'trekresearch-loop-scope', `${sessionId}.json`),
|
||
JSON.stringify({ runId: RUN_ID, startedAt: startedAt.toISOString() }),
|
||
);
|
||
const lines = Array.from({ length: turns }, (_, i) =>
|
||
JSON.stringify({ ts: new Date().toISOString(), runId: RUN_ID, dimension: `d${i}`, effort: 'high', slot: i + 1 }),
|
||
);
|
||
// The tombstone research-loop-cap.mjs appends when it denies a turn for
|
||
// budget. Its presence is what tells this hook "the gate already said no".
|
||
if (exhausted) lines.push(JSON.stringify({ ts: new Date().toISOString(), runId: RUN_ID, exhausted: true }));
|
||
writeFileSync(join(dir, 'trekresearch-loop-ledger.jsonl'), lines.length ? lines.join('\n') + '\n' : '');
|
||
return dir;
|
||
}
|
||
|
||
// TREKRESEARCH_MAX_CONV_TURNS=1 => budget = 1 * MAX_TOTAL_DIMENSIONS (8).
|
||
const CAPPED_ENV = { VOYAGE_STORM_ENABLED: '1', TREKRESEARCH_MAX_CONV_TURNS: '1' };
|
||
const BUDGET = 8;
|
||
|
||
function searchInput(sessionId = SESSION) {
|
||
return {
|
||
session_id: sessionId,
|
||
hook_event_name: 'PreToolUse',
|
||
tool_name: 'WebSearch',
|
||
tool_input: { query: 'claude code hooks reference' },
|
||
agent_id: 'aa6d19525a4680fe0',
|
||
agent_type: 'general-purpose',
|
||
};
|
||
}
|
||
|
||
// -----------------------------------------------------------------------
|
||
// DENY — budget spent
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap DENIES once the budget gate has denied a turn (tombstone present)', async () => {
|
||
const dir = fixture({ turns: BUDGET, exhausted: true });
|
||
const { code, stderr } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 2);
|
||
assert.match(stderr, /loop cap/i, 'stderr must name the cap it enforced');
|
||
assert.match(stderr, new RegExp(`${BUDGET}`), 'stderr must state the budget');
|
||
});
|
||
|
||
test('pre-agent-cap DENIES above the budget too (a breached ledger, whatever caused it)', async () => {
|
||
const dir = fixture({ turns: BUDGET + 5 });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 2);
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// ALLOW — under the cap, and ON the cap.
|
||
//
|
||
// allowTurn appends BEFORE the turn runs, so during the FINAL granted turn the
|
||
// ledger already holds `budget` records. Denying at `used >= budget` therefore
|
||
// blocked that turn's own tool calls: the primitive granted B turns, the
|
||
// harness permitted B-1, and an exhausted run always ended through an exit-2
|
||
// denial rather than the graceful "cap exhausted" exit the prose defines. The
|
||
// boundary belongs one turn later, and the tombstone above — not the count —
|
||
// is what marks a run actually finished.
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap ALLOWS a loop turn under the budget', async () => {
|
||
const dir = fixture({ turns: BUDGET - 1 });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
test('pre-agent-cap ALLOWS the FINAL granted turn — its own record is already on the ledger', async () => {
|
||
const dir = fixture({ turns: BUDGET });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(
|
||
code, 0,
|
||
'turn B is granted and in flight; denying it makes the harness permit B-1 turns and forces the wrong exit',
|
||
);
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// DOES NOT OVER-BLOCK — the property that keeps this hook safe to wire
|
||
// globally. A broken PreToolUse hook would brick every session on the box.
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap ALLOWS an unrelated session even when a loop is exhausted', async () => {
|
||
const dir = fixture({ turns: BUDGET });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput('some-other-session'), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0, 'no scope marker for this session_id => out of scope');
|
||
});
|
||
|
||
test('pre-agent-cap ALLOWS when no scope marker directory exists at all', async () => {
|
||
const dir = mkdtempSync(join(tmpdir(), 'voyage-cap-empty-'));
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
test('pre-agent-cap ALLOWS on unparsable stdin', async () => {
|
||
const dir = fixture({ turns: BUDGET });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, 'not json at all', {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
test('pre-agent-cap ALLOWS when the input carries no session_id', async () => {
|
||
const dir = fixture({ turns: BUDGET });
|
||
const input = searchInput();
|
||
delete input.session_id;
|
||
const { code } = await runHookWithEnv(CAP_HOOK, input, {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// TTL / auto-reset — a marker left behind by a crashed run must not deny
|
||
// tool calls forever.
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap ALLOWS when the scope marker is older than the TTL', async () => {
|
||
const dir = fixture({ turns: BUDGET, startedAt: new Date(Date.now() - 48 * 3600 * 1000) });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
VOYAGE_CAP_SCOPE_TTL_MS: '1000',
|
||
});
|
||
assert.strictEqual(code, 0, 'a stale marker must auto-reset, not deny forever');
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// Fail CLOSED once in scope — the hook's own header says a budget control
|
||
// that cannot count must not grant. The unreadable-ledger branch returned 0
|
||
// and therefore ALLOWED, which is the opposite. A directory standing where
|
||
// the ledger file belongs reproduces it portably (EISDIR).
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap DENIES when the ledger cannot be read at all (fail closed)', async () => {
|
||
const dir = fixture({ turns: 0 });
|
||
const ledgerPath = join(dir, 'trekresearch-loop-ledger.jsonl');
|
||
rmSync(ledgerPath, { force: true });
|
||
mkdirSync(ledgerPath, { recursive: true });
|
||
const { code, stderr } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 2, 'an in-scope run whose ledger cannot be counted must not be granted');
|
||
assert.match(stderr, /could not be read|unreadable/i, 'stderr must say counting failed, not that the budget is spent');
|
||
});
|
||
|
||
test('pre-agent-cap ALLOWS an unreadable ledger when the session is OUT of scope', async () => {
|
||
const dir = fixture({ turns: 0, sessionId: 'a-different-session' });
|
||
const ledgerPath = join(dir, 'trekresearch-loop-ledger.jsonl');
|
||
rmSync(ledgerPath, { force: true });
|
||
mkdirSync(ledgerPath, { recursive: true });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0, 'fail-closed is scoped to the loop, it must not brick unrelated sessions');
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// Kill switch + default-off
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap kill switch VOYAGE_DISABLE_CAP_HOOK=1 allows an exhausted loop', async () => {
|
||
const dir = fixture({ turns: BUDGET });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
VOYAGE_DISABLE_CAP_HOOK: '1',
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
test('pre-agent-cap is inert when VOYAGE_STORM_ENABLED is not 1 (default-off)', async () => {
|
||
const dir = fixture({ turns: BUDGET });
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
VOYAGE_STORM_ENABLED: '0',
|
||
CLAUDE_PLUGIN_DATA: dir,
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// The measured environment — CLAUDE_PLUGIN_DATA is EMPTY in the Bash tool
|
||
// env, so that is the environment every real run happens in. The writer (the
|
||
// Phase 5 bash snippet) and the reader (this hook) must land on the SAME
|
||
// fallback root, or the hook allows unconditionally while claiming to enforce.
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap enforces via the fallback root when CLAUDE_PLUGIN_DATA is absent', async () => {
|
||
const home = mkdtempSync(join(tmpdir(), 'voyage-cap-home-'));
|
||
const root = join(home, '.claude', 'voyage');
|
||
mkdirSync(join(root, 'trekresearch-loop-scope'), { recursive: true });
|
||
writeFileSync(
|
||
join(root, 'trekresearch-loop-scope', `${SESSION}.json`),
|
||
JSON.stringify({ runId: RUN_ID, startedAt: new Date().toISOString() }),
|
||
);
|
||
writeFileSync(
|
||
join(root, 'trekresearch-loop-ledger.jsonl'),
|
||
[
|
||
...Array.from({ length: BUDGET }, (_, i) =>
|
||
JSON.stringify({ ts: new Date().toISOString(), runId: RUN_ID, dimension: `d${i}`, effort: 'high', slot: i + 1 })),
|
||
JSON.stringify({ ts: new Date().toISOString(), runId: RUN_ID, exhausted: true }),
|
||
].join('\n') + '\n',
|
||
);
|
||
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: '',
|
||
HOME: home,
|
||
});
|
||
assert.strictEqual(code, 2, 'the hook must find marker AND ledger under the fallback root and deny');
|
||
});
|
||
|
||
test('pre-agent-cap allows under budget in the fallback root — the fallback is not a blanket deny', async () => {
|
||
const home = mkdtempSync(join(tmpdir(), 'voyage-cap-home-'));
|
||
const root = join(home, '.claude', 'voyage');
|
||
mkdirSync(join(root, 'trekresearch-loop-scope'), { recursive: true });
|
||
writeFileSync(
|
||
join(root, 'trekresearch-loop-scope', `${SESSION}.json`),
|
||
JSON.stringify({ runId: RUN_ID, startedAt: new Date().toISOString() }),
|
||
);
|
||
writeFileSync(join(root, 'trekresearch-loop-ledger.jsonl'), '');
|
||
|
||
const { code } = await runHookWithEnv(CAP_HOOK, searchInput(), {
|
||
...CAPPED_ENV,
|
||
CLAUDE_PLUGIN_DATA: '',
|
||
HOME: home,
|
||
});
|
||
assert.strictEqual(code, 0);
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// Append-only counting — the hook reads the ledger, it never writes it.
|
||
// Writing per tool call would make the cap count its own enforcement.
|
||
// -----------------------------------------------------------------------
|
||
test('pre-agent-cap never writes to the ledger', async () => {
|
||
const dir = fixture({ turns: 2 });
|
||
const ledgerPath = join(dir, 'trekresearch-loop-ledger.jsonl');
|
||
const before = readFileSync(ledgerPath, 'utf-8');
|
||
await runHookWithEnv(CAP_HOOK, searchInput(), { ...CAPPED_ENV, CLAUDE_PLUGIN_DATA: dir });
|
||
assert.strictEqual(readFileSync(ledgerPath, 'utf-8'), before);
|
||
});
|
||
|
||
// -----------------------------------------------------------------------
|
||
// Wiring — pattern from tests/hooks/hooks-json-stop-wired.test.mjs
|
||
// -----------------------------------------------------------------------
|
||
function invocationOf(h) {
|
||
return [h.command || '', ...(h.args || [])].join(' ').trim();
|
||
}
|
||
|
||
test('hooks.json wires pre-agent-cap.mjs on PreToolUse with ${CLAUDE_PLUGIN_ROOT}', () => {
|
||
const cfg = JSON.parse(readFileSync(HOOKS_JSON, 'utf8'));
|
||
const invocations = (cfg.hooks.PreToolUse || []).flatMap((entry) =>
|
||
(entry.hooks || []).map(invocationOf),
|
||
);
|
||
const capInvocation = invocations.find((cmd) => cmd.includes('pre-agent-cap.mjs'));
|
||
assert.ok(capInvocation, `no PreToolUse hook references pre-agent-cap.mjs. Found: ${JSON.stringify(invocations)}`);
|
||
assert.match(capInvocation, /\$\{CLAUDE_PLUGIN_ROOT\}/, 'relative paths fail in headless sessions');
|
||
assert.match(capInvocation, /^node\s+/);
|
||
});
|
||
|
||
test('hooks.json matcher for pre-agent-cap covers the loop’s outbound surface', () => {
|
||
const cfg = JSON.parse(readFileSync(HOOKS_JSON, 'utf8'));
|
||
const entry = (cfg.hooks.PreToolUse || []).find((e) =>
|
||
(e.hooks || []).some((h) => invocationOf(h).includes('pre-agent-cap.mjs')),
|
||
);
|
||
assert.ok(entry, 'pre-agent-cap entry missing from PreToolUse');
|
||
for (const tool of ['WebSearch', 'WebFetch', 'Task']) {
|
||
assert.match(entry.matcher, new RegExp(tool), `matcher must cover ${tool}`);
|
||
}
|
||
});
|