voyage/tests/lib/criteria-runner.test.mjs
Kjell Tore Guttormsen c23b009738
fix(review): run the success-criteria commands and hand the reviewer the result (D-04)
The rubric required `brief-conformance-reviewer` to classify a Success
Criterion as Full only when "its verification command/test exists and passes".
Its tools are `Read`, `Glob`, `Grep`. It cannot run anything, so "passes" was
either guessed from the command's mere existence or quietly downgraded to
"exists" — a BLOCKER-tier rule key resting on an impression.

The reviewer stays read-only — a reviewer that executes the code it reviews is
not an independent reviewer. The command does the running instead:

- `/trekreview` Phase 4.5 runs the brief's `## Success Criteria` commands
  through `lib/verification/criteria-runner.mjs --brief --evidence` and captures
  the block as `sc_evidence_block`, pasted verbatim into the reviewer prompt in
  Phase 5. The exit code does not stop the review — a failing criterion is
  exactly what the review exists to find.
- `formatCriteriaEvidence` builds that block in code: one row per criterion with
  the command, the exit code and the first output line. Chose a code-built block
  over an orchestrator-written summary so the orchestrator cannot narrate a pass
  that never happened.
- The rubric now judges the supplied result: `PASS` supports Full, `FAILED` /
  `BLOCKED` is `Broken` with the exit code cited, and `NOT RUN` is the absence
  of a measurement — never evidence in either direction.
- Phase 4.5 is skipped in `quick` mode: that mode does not launch the
  conformance reviewer, so there is nobody to hand the result to.

Red first: seven tests in `tests/lib/criteria-runner.test.mjs` against a
committed brief fixture whose three criteria pass, fail, and are prose-only.
The two doc pins were verified red against the pre-fix files (rubric asked
"exists and passes"; no Phase 4.5; the block reached nobody).

Suite: 1117 (1115/0/2), up 9.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-18 01:34:18 +02:00

352 lines
14 KiB
JavaScript

// tests/lib/criteria-runner.test.mjs
// The criteria runner is what makes a declared check a RUN check: it parses the
// falsifiable criteria a plan (`## Verification`) or a brief (`## Success
// Criteria`) declares, screens each command through the executor's own
// PreToolUse denylist, runs it, and returns a verdict built from exit codes.
//
// Fail-closed is the whole point: a criterion that cannot run (placeholder, no
// command, screen unavailable) must never read as "passed".
import { test } from 'node:test';
import { strict as assert } from 'node:assert';
import { spawnSync } from 'node:child_process';
import { join, dirname } from 'node:path';
import { fileURLToPath } from 'node:url';
import { mkdtempSync, writeFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import {
parsePlanVerification,
parseSuccessCriteria,
screenCommand,
runCriteria,
summarize,
runPlanVerification,
runSuccessCriteriaChecks,
formatCriteriaEvidence,
render,
} from '../../lib/verification/criteria-runner.mjs';
const HERE = dirname(fileURLToPath(import.meta.url));
const ROOT = join(HERE, '..', '..');
const CLI = join(ROOT, 'lib', 'verification', 'criteria-runner.mjs');
const FIX = join(ROOT, 'tests', 'fixtures');
const HOOK = join(ROOT, 'hooks', 'scripts', 'pre-bash-executor.mjs');
// An exec double: maps a command string to {status, stdout, stderr}.
function execDouble(table) {
const calls = [];
const exec = (command) => {
calls.push(command);
return table[command] ?? { status: 127, stdout: '', stderr: 'not in table' };
};
exec.calls = calls;
return exec;
}
// A screen double that allows everything, so exec behaviour can be tested alone.
const allowAll = () => ({ allowed: true, rule: '' });
// --- parsing ---------------------------------------------------------------
test('parsePlanVerification: checkbox bullets become V1..Vn with their command', () => {
const md = [
'# Plan',
'',
'## Verification',
'',
'- [ ] `npm test` -> expected: exit 0',
'- [x] `node --test tests/lib/x.test.mjs` -> expected: 3 passing',
'',
'## Estimated Scope',
'',
'- [ ] `not-a-criterion` (outside the section)',
].join('\n');
const criteria = parsePlanVerification(md);
assert.equal(criteria.length, 2);
assert.deepEqual(criteria.map((c) => c.label), ['V1', 'V2']);
assert.equal(criteria[0].command, 'npm test');
assert.equal(criteria[1].command, 'node --test tests/lib/x.test.mjs');
assert.match(criteria[0].text, /expected: exit 0/);
});
test('parsePlanVerification: a template placeholder is unrunnable, not a command', () => {
const md = '## Verification\n\n- [ ] `{exact command}` -> expected: `{exact output}`\n';
const criteria = parsePlanVerification(md);
assert.equal(criteria.length, 1);
assert.equal(criteria[0].command, null);
assert.equal(criteria[0].reason, 'placeholder');
});
test('parsePlanVerification: a missing section yields no criteria', () => {
assert.deepEqual(parsePlanVerification('# Plan\n\n## Steps\n\n- do a thing\n'), []);
});
test('parseSuccessCriteria: bullets become SC1..SCn and take the FIRST backticked span', () => {
const md = [
'## Success Criteria',
'',
'- All existing tests pass: `npm test` exits 0',
'- Endpoint returns 200: `curl -s localhost:3000/health` -> `"ok"`',
'- No new runtime dependencies are introduced',
'',
'## Research Plan',
].join('\n');
const criteria = parseSuccessCriteria(md);
assert.deepEqual(criteria.map((c) => c.label), ['SC1', 'SC2', 'SC3']);
assert.equal(criteria[0].command, 'npm test');
assert.equal(criteria[1].command, 'curl -s localhost:3000/health');
assert.equal(criteria[2].command, null);
assert.equal(criteria[2].reason, 'no-command');
});
// --- screening -------------------------------------------------------------
test('screenCommand: the real executor denylist blocks a catastrophic command', () => {
const verdict = screenCommand('rm -rf ~', { hookPath: HOOK });
assert.equal(verdict.allowed, false);
assert.match(verdict.rule, /rm -rf|destruction/i);
});
test('screenCommand: an ordinary command passes the real denylist', () => {
assert.equal(screenCommand('npm test', { hookPath: HOOK }).allowed, true);
});
test('screenCommand: an unavailable screen denies (fail-closed, never silently allows)', () => {
const verdict = screenCommand('npm test', { hookPath: join(ROOT, 'hooks', 'scripts', 'no-such-hook.mjs') });
assert.equal(verdict.allowed, false);
assert.match(verdict.rule, /screen unavailable/i);
});
// --- running ---------------------------------------------------------------
test('runCriteria: exit 0 passes, a non-zero exit fails, output is captured', () => {
const criteria = parsePlanVerification(
'## Verification\n\n- [ ] `good`\n- [ ] `bad`\n'
);
const exec = execDouble({
good: { status: 0, stdout: 'all green\n', stderr: '' },
bad: { status: 1, stdout: '', stderr: '1 failing\n' },
});
const results = runCriteria(criteria, { exec, screen: allowAll });
assert.deepEqual(results.map((r) => r.status), ['passed', 'failed']);
assert.equal(results[0].exitCode, 0);
assert.equal(results[1].exitCode, 1);
assert.match(results[1].output, /1 failing/);
assert.deepEqual(exec.calls, ['good', 'bad']);
});
test('runCriteria: a blocked command is marked blocked and is NEVER executed', () => {
const criteria = parsePlanVerification('## Verification\n\n- [ ] `rm -rf ~`\n');
const exec = execDouble({});
const results = runCriteria(criteria, {
exec,
screen: () => ({ allowed: false, rule: 'Filesystem root/home destruction' }),
});
assert.equal(results[0].status, 'blocked');
assert.equal(results[0].exitCode, null);
assert.match(results[0].output, /Filesystem root\/home destruction/);
assert.deepEqual(exec.calls, [], 'a blocked command must not reach the shell');
});
test('runCriteria: a criterion with no command is unrunnable, not passed', () => {
const criteria = parseSuccessCriteria('## Success Criteria\n\n- No new dependencies\n');
const results = runCriteria(criteria, { exec: execDouble({}), screen: allowAll });
assert.equal(results[0].status, 'unrunnable');
assert.equal(results[0].exitCode, null);
});
test('runCriteria: output is capped so a verbose command cannot flood a prompt', () => {
const criteria = parsePlanVerification('## Verification\n\n- [ ] `loud`\n');
const exec = execDouble({ loud: { status: 0, stdout: 'x'.repeat(10000), stderr: '' } });
const results = runCriteria(criteria, { exec, screen: allowAll, maxOutput: 200 });
assert.ok(results[0].output.length < 400, `capped, got ${results[0].output.length}`);
assert.match(results[0].output, /truncated/);
});
// --- the verdict -----------------------------------------------------------
test('summarize: one failing criterion makes the run NOT ok', () => {
const results = [
{ status: 'passed' }, { status: 'failed' }, { status: 'passed' },
];
const s = summarize(results, { requireCommand: true });
assert.equal(s.ok, false);
assert.equal(s.failed, 1);
assert.equal(s.passed, 2);
assert.equal(s.total, 3);
});
test('summarize: in plan mode an unrunnable criterion makes the run NOT ok', () => {
const s = summarize([{ status: 'passed' }, { status: 'unrunnable' }], { requireCommand: true });
assert.equal(s.ok, false);
assert.equal(s.unrunnable, 1);
});
test('summarize: in brief mode an unrunnable criterion is reported, not failed', () => {
const s = summarize([{ status: 'passed' }, { status: 'unrunnable' }], { requireCommand: false });
assert.equal(s.ok, true);
assert.equal(s.unrunnable, 1);
});
test('summarize: a blocked criterion is never ok, in either mode', () => {
for (const requireCommand of [true, false]) {
assert.equal(summarize([{ status: 'blocked' }], { requireCommand }).ok, false);
}
});
test('summarize: zero criteria is NOT ok in plan mode (a plan that promises nothing)', () => {
assert.equal(summarize([], { requireCommand: true }).ok, false);
});
// --- the single-session path, end to end -----------------------------------
test('runPlanVerification: a plan whose success criterion FAILS fells the run', () => {
const report = runPlanVerification(join(FIX, 'plan-verification-fails.md'));
assert.equal(report.kind, 'plan');
assert.equal(report.summary.ok, false);
assert.equal(report.summary.failed, 1);
const failed = report.results.find((r) => r.status === 'failed');
assert.ok(failed, 'the failing criterion is reported by label');
assert.match(failed.label, /^V\d+$/);
});
test('runPlanVerification: a plan whose criteria all pass is ok', () => {
const report = runPlanVerification(join(FIX, 'plan-verification-passes.md'));
assert.equal(report.summary.ok, true);
assert.equal(report.summary.failed, 0);
assert.equal(report.summary.total, 2);
});
test('runPlanVerification: a plan with no ## Verification section is NOT ok', () => {
const dir = mkdtempSync(join(tmpdir(), 'criteria-runner-'));
const p = join(dir, 'plan.md');
writeFileSync(p, '# Plan\n\n## Steps\n\n- do a thing\n');
const report = runPlanVerification(p);
assert.equal(report.summary.ok, false);
assert.equal(report.error.code, 'NO_VERIFICATION_SECTION');
});
test('render: the report names every non-passing criterion and its exit code', () => {
const report = runPlanVerification(join(FIX, 'plan-verification-fails.md'));
const text = render(report);
assert.match(text, /FAILED/);
assert.match(text, /exit 1/);
});
// --- the CLI ---------------------------------------------------------------
function cli(args) {
return spawnSync(process.execPath, [CLI, ...args], { encoding: 'utf8', cwd: ROOT });
}
test('CLI: --plan exits 1 when a criterion fails', () => {
const r = cli(['--plan', join(FIX, 'plan-verification-fails.md')]);
assert.equal(r.status, 1, r.stderr);
assert.match(r.stdout, /FAILED/);
});
test('CLI: --plan exits 0 when every criterion passes', () => {
const r = cli(['--plan', join(FIX, 'plan-verification-passes.md')]);
assert.equal(r.status, 0, r.stderr + r.stdout);
});
test('CLI: --json emits a parseable report with the summary', () => {
const r = cli(['--plan', join(FIX, 'plan-verification-fails.md'), '--json']);
assert.equal(r.status, 1);
const out = JSON.parse(r.stdout);
assert.equal(out.summary.ok, false);
assert.equal(out.results.length, out.summary.total);
});
test('CLI: a missing file exits 2 — a read error is not a failed criterion', () => {
const r = cli(['--plan', join(FIX, 'no-such-plan.md')]);
assert.equal(r.status, 2);
assert.match(r.stderr, /criteria-runner/);
});
test('CLI: an unknown argument exits 2 with usage', () => {
const r = cli(['--nope']);
assert.equal(r.status, 2);
assert.match(r.stderr, /usage/);
});
test('CLI: no mode flag exits 2 — it never guesses which artifact it was given', () => {
const r = cli([]);
assert.equal(r.status, 2);
assert.match(r.stderr, /usage/);
});
// --- the brief path: evidence handed to a read-only reviewer ----------------
//
// D-04: brief-conformance-reviewer is asked to judge whether a Success
// Criterion's verification command "exists and passes", but its tools are
// Read/Glob/Grep — it cannot run anything. /trekreview runs the commands and
// hands over the RESULT, and the block it hands over is built by code so the
// orchestrator cannot narrate a pass that never happened.
test('runSuccessCriteriaChecks: each criterion gets a real result, prose ones are NOT RUN', () => {
const report = runSuccessCriteriaChecks(join(FIX, 'brief-success-criteria.md'));
assert.equal(report.kind, 'brief');
assert.deepEqual(report.results.map((r) => r.label), ['SC1', 'SC2', 'SC3']);
assert.equal(report.results[0].status, 'passed');
assert.equal(report.results[1].status, 'failed');
assert.equal(report.results[1].exitCode, 1);
assert.equal(report.results[2].status, 'unrunnable');
assert.equal(report.summary.ok, false, 'a failing criterion is never ok, in either mode');
});
test('runSuccessCriteriaChecks: a brief with no Success Criteria section reports the code', () => {
const dir = mkdtempSync(join(tmpdir(), 'criteria-runner-'));
const p = join(dir, 'brief.md');
writeFileSync(p, '# Brief\n\n## Goal\n\nSomething.\n');
const report = runSuccessCriteriaChecks(p);
assert.equal(report.error.code, 'NO_SUCCESS_CRITERIA_SECTION');
assert.deepEqual(report.results, []);
});
test('formatCriteriaEvidence: one row per criterion, with command and exit code', () => {
const report = runSuccessCriteriaChecks(join(FIX, 'brief-success-criteria.md'));
const block = formatCriteriaEvidence(report);
for (const label of ['SC1', 'SC2', 'SC3']) assert.match(block, new RegExp(`\\| ${label} \\|`));
assert.match(block, /PASS/);
assert.match(block, /FAILED/);
assert.match(block, /NOT RUN/);
assert.match(block, /exit 1/);
});
test('formatCriteriaEvidence: the block forbids inferring a pass for a criterion with no result', () => {
const report = runSuccessCriteriaChecks(join(FIX, 'brief-success-criteria.md'));
const block = formatCriteriaEvidence(report);
assert.match(block, /NOT RUN/);
assert.match(
block, /never.*(infer|assume)/i,
'the evidence block must state in-band that NOT RUN is not a pass — the reviewer cannot re-run it',
);
});
test('formatCriteriaEvidence: an unreadable section still yields a block that says so', () => {
const dir = mkdtempSync(join(tmpdir(), 'criteria-runner-'));
const p = join(dir, 'brief.md');
writeFileSync(p, '# Brief\n\n## Goal\n\nSomething.\n');
const block = formatCriteriaEvidence(runSuccessCriteriaChecks(p));
assert.match(block, /NO_SUCCESS_CRITERIA_SECTION/);
assert.match(block, /no criteria were run/i);
});
test('CLI: --evidence emits the reviewer block, and --brief still exits 1 on a failure', () => {
const r = cli(['--brief', join(FIX, 'brief-success-criteria.md'), '--evidence']);
assert.equal(r.status, 1, r.stderr);
assert.match(r.stdout, /\| SC1 \|/);
assert.match(r.stdout, /NOT RUN/);
});
test('CLI: --evidence and --json together exit 2 — one output shape at a time', () => {
const r = cli(['--brief', join(FIX, 'brief-success-criteria.md'), '--evidence', '--json']);
assert.equal(r.status, 2);
assert.match(r.stderr, /usage/);
});