config-audit/tests/scanners/posture-humanizer.test.mjs
Kjell Tore Guttormsen 7a794b47eb fix(scanners)!: a finding ID names the check, not the emission (M-BUG-28)
BREAKING CHANGE: the {NNN} in CA-{SCANNER}-{NNN} identifies the check that
produced the finding. It used to be the finding's position in that scanner's
output for that run, which made it unstable across CONFIGURATIONS, not just
across releases as STATE framed it. Measured on two fixtures: "No custom
subagents" was CA-GAP-007 on minimal-project and CA-GAP-004 on healthy-project.
A user who fixed an unrelated earlier gap silently renumbered every later one,
so a .config-audit-ignore pin retargeted to a neighbouring finding with no
version change at all.

Second measured arm: README already documented the opposite scheme. It and the
scanner headers describe ~20 numbers as check codes (CA-SKL-003 = oversized
body, CA-PLH-015 = folder shadowing, CA-TOK-006 = schema deferral), and the
counter could only produce those in the all-fire case -- source-order positions
are 4, 3 and 8. The documentation described the scheme; the implementation was
what was wrong. Every published number is preserved by construction and pinned
exhaustively in tests/lib/finding-codes.test.mjs.

scanners/lib/finding-codes.mjs is the single authority. Every finding() call
passes a `code`; an undeclared or missing one THROWS. No counter fallback --
that would reproduce D1's findGapId -> 'unknown' silent degradation and let a
half-converted scanner ship IDs that look valid. findingCounter/resetCounter
are deleted outright, not left as no-ops. Retirement is now a mechanism:
RETIRED_CODES tombstones a withdrawn key so its number is never reissued,
seeded with GAP t3_8 -- the D1 removal that opened this chunk.

IDs are consequently NOT unique per finding: one check failing in three files
emits three findings sharing an ID. That inverts which consumer is correct, so
every f.id/findingId site was classified before the change. diff-engine and
most of fix-engine already keyed on scanner+title+file (drift was never lying);
fix-engine's verification did not, and keyed on the ID alone -- fixing one of
two sibling instances marked both fixed, and the untouched one, still present
in the re-scan, was reported as a REGRESSION. Red test first, then keyed on
(findingId, file), which both planFixes and applyFixes already carry.
plugin-health's crossIds Set was measured and is a clean negative: cross
findings are allFindings.slice(crossPluginStart) and codes 18/19 are emitted
only in that tail, so the partition holds by construction.

unknownSuppressions() reports a pin that names no declared check, in the
--output-file payload (ux-rules rule 2 -- a stderr-only warning is invisible to
the commands) and only when one exists, so a clean config is byte-identical.
That is what makes the break safe: a stale pin goes loud instead of dying quiet.

Frozen tests/snapshots/v5.0.0/ untouched on disk. IDs are masked out of that
comparison (mask-finding-ids.mjs) rather than re-derived -- re-deriving
positional IDs would assert the retired scheme against itself, and #58's
isGapEntry off-by-one is the measured example of that misfiring. The dead
re-derivation is removed from strip-retired-gap.mjs. default-output snapshots
re-approved after confirming the diff is IDs and nothing else.

Guards, each seen red against its own defect: a missing code (scanner errors
out mid-sweep), an orphan declaration, a resurrected retired key, and a
documented ID naming no check. The sweep asserts the union across all 16
scanners, never per scanner -- a per-scanner assertion goes green on a partial
conversion.

Fasit written before implementation: docs/mbug28-id-semantics-fasit.local.md,
including one correction made before running (CML has 12 checks over 13 call
sites -- the anchored and calibrated char-budget arms are one check, which a
repeated-title sweep found and my call-site count had missed).

Suite 1535 -> 1573, 0 failing.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MyqCQKK2ornJ1jFWwqx17E
2026-08-09 23:26:36 +02:00

187 lines
8.2 KiB
JavaScript

import { describe, it } from 'node:test';
import assert from 'node:assert/strict';
import { resolve, dirname, join } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';
import { execFile } from 'node:child_process';
import { promisify } from 'node:util';
import { readFile, unlink } from 'node:fs/promises';
import { hermeticEnv } from '../helpers/hermetic-home.mjs';
import { stripHotspotLoadPattern } from '../helpers/strip-hotspot-load-pattern.mjs';
import { stripAddedScanners, stripAddedScannerStderr } from '../helpers/strip-added-scanner.mjs';
import { stripRetiredGap, maskGapTallyStderr } from '../helpers/strip-retired-gap.mjs';
import { maskFindingIds } from '../helpers/mask-finding-ids.mjs';
const exec = promisify(execFile);
const __dirname = dirname(fileURLToPath(import.meta.url));
const REPO = resolve(__dirname, '../..');
const CLI = resolve(REPO, 'scanners/posture.mjs');
const FIXTURE = resolve(REPO, 'tests/fixtures/marketplace-medium');
const POSTURE_JSON_SNAPSHOT = resolve(REPO, 'tests/snapshots/v5.0.0/posture.json');
const POSTURE_STDERR_SNAPSHOT = resolve(REPO, 'tests/snapshots/v5.0.0-stderr/posture.txt');
/**
* Normalize a runPosture result for snapshot comparison by zeroing out
* time-varying fields, machine-specific paths, and ancestor-cascade-derived
* counts. `claudeMdEstimatedTokens` reflects walkClaudeMdCascade walking
* upward from the fixture; any docs edit to this plugin's own CLAUDE.md
* ripples into it even though scanner behavior is unchanged.
*/
function normalizePosture(p) {
const out = JSON.parse(JSON.stringify(p));
if (out.scannerEnvelope) {
if (out.scannerEnvelope.meta) {
out.scannerEnvelope.meta.target = '<TARGET>';
out.scannerEnvelope.meta.timestamp = '<TIMESTAMP>';
}
if (Array.isArray(out.scannerEnvelope.scanners)) {
for (const s of out.scannerEnvelope.scanners) {
s.duration_ms = 0;
if (s.activeConfig && 'claudeMdEstimatedTokens' in s.activeConfig) {
s.activeConfig.claudeMdEstimatedTokens = '<ANCESTOR_DERIVED>';
}
}
}
}
return maskFindingIds(stripRetiredGap(stripAddedScanners(stripHotspotLoadPattern(out))));
}
/**
* Strip time-varying durations (Xms) so progress lines compare verbatim across
* runs, and drop the additive [OST] progress line (v5.6 C) so a 14-scanner live
* run matches the frozen 13-scanner v5.0.0 stderr scorecard.
*/
function normalizeStderr(s) {
return maskGapTallyStderr(stripAddedScannerStderr(s)).replace(/\(\d+ms\)/g, '(0ms)');
}
async function runPosture(flags) {
const proc = await exec('node', [CLI, FIXTURE, ...flags], {
timeout: 60000,
cwd: REPO,
env: hermeticEnv(),
}).catch(err => err); // posture exits non-zero on findings — capture either way
return {
stdout: proc.stdout || '',
stderr: proc.stderr || '',
};
}
describe('posture humanizer wiring (Step 6)', () => {
describe('--json mode (SC-6: byte-equal stdout)', () => {
it('stdout JSON deepEquals v5.0.0 snapshot', async () => {
const { stdout } = await runPosture(['--json']);
const actual = JSON.parse(stdout);
const expected = JSON.parse(await readFile(POSTURE_JSON_SNAPSHOT, 'utf-8'));
assert.deepStrictEqual(normalizePosture(actual), normalizePosture(expected));
});
it('does NOT write a scorecard to stderr (suppressed)', async () => {
const { stderr } = await runPosture(['--json']);
assert.ok(!stderr.includes('Config-Audit Health Score'),
'stderr must NOT contain scorecard in --json mode');
assert.ok(!stderr.includes('Configuration health'),
'stderr must NOT contain humanized scorecard in --json mode');
});
it('preserves v5.0.0 finding shape (no humanizer fields in scannerEnvelope)', async () => {
const { stdout } = await runPosture(['--json']);
const actual = JSON.parse(stdout);
for (const s of actual.scannerEnvelope.scanners) {
for (const f of s.findings) {
assert.equal(f.userImpactCategory, undefined,
`${f.id}: --json findings must not have userImpactCategory`);
}
}
});
});
describe('--raw mode (SC-7: byte-equal stdout + verbatim stderr)', () => {
it('stdout JSON deepEquals v5.0.0 snapshot', async () => {
const { stdout } = await runPosture(['--raw']);
const actual = JSON.parse(stdout);
const expected = JSON.parse(await readFile(POSTURE_JSON_SNAPSHOT, 'utf-8'));
assert.deepStrictEqual(normalizePosture(actual), normalizePosture(expected));
});
it('stderr scorecard verbatim matches v5.0.0 stderr snapshot', async () => {
const { stderr } = await runPosture(['--raw']);
const expected = await readFile(POSTURE_STDERR_SNAPSHOT, 'utf-8');
// Compare the scorecard portion verbatim (modulo timing in scanner progress lines)
assert.equal(normalizeStderr(stderr).trim(), normalizeStderr(expected).trim());
});
it('preserves v5.0.0 finding shape in stdout', async () => {
const { stdout } = await runPosture(['--raw']);
const actual = JSON.parse(stdout);
for (const s of actual.scannerEnvelope.scanners) {
for (const f of s.findings) {
assert.equal(f.userImpactCategory, undefined,
`${f.id}: --raw findings must not have userImpactCategory`);
}
}
});
});
describe('default mode (humanized scorecard)', () => {
it('writes humanized scorecard to stderr', async () => {
const { stderr } = await runPosture([]);
// Humanized scorecard must contain at least one user-friendly cue not in raw v5.0.0
const hasGradeContext = /healthy|good shape|attention|polish|setup/i.test(stderr);
assert.ok(hasGradeContext,
`humanized stderr scorecard must contain user-friendly phrasing, got:\n${stderr}`);
});
it('does NOT write JSON to stdout in default mode', async () => {
const { stdout } = await runPosture([]);
assert.equal(stdout.trim(), '', 'default mode must not write JSON to stdout');
});
it('humanized scorecard differs byte-wise from v5.0.0 stderr', async () => {
const { stderr } = await runPosture([]);
const expected = await readFile(POSTURE_STDERR_SNAPSHOT, 'utf-8');
assert.notEqual(normalizeStderr(stderr).trim(), normalizeStderr(expected).trim(),
'humanized stderr must differ from v5.0.0 verbatim stderr');
});
});
// M-BUG-12: feature-gap.md/posture.md read findings from posture.mjs --output-file
// and group on humanizer fields (userActionLanguage etc.). Default-mode output-file
// must therefore humanize findings inside the nested scannerEnvelope; --raw stays raw.
describe('default mode --output-file (M-BUG-12: humanized findings)', () => {
it('writes humanized GAP findings (userActionLanguage defined) to the output file', async () => {
const tmp = join(tmpdir(), `ca-posture-outfile-${process.pid}.json`);
try {
await runPosture(['--output-file', tmp]);
const env = JSON.parse(await readFile(tmp, 'utf-8'));
const gap = env.scannerEnvelope.scanners.find(s => s.scanner === 'GAP');
assert.ok(gap, 'GAP scanner must be present in the output file');
assert.ok(gap.findings.length > 0, 'fixture must yield at least one GAP finding');
for (const f of gap.findings) {
assert.notEqual(f.userActionLanguage, undefined,
`${f.id}: default-mode output-file findings must carry userActionLanguage`);
assert.notEqual(f.userImpactCategory, undefined,
`${f.id}: default-mode output-file findings must carry userImpactCategory`);
}
} finally {
await unlink(tmp).catch(() => {});
}
});
it('--raw --output-file keeps v5.0.0 raw finding shape (no humanizer fields)', async () => {
const tmp = join(tmpdir(), `ca-posture-outfile-raw-${process.pid}.json`);
try {
await runPosture(['--raw', '--output-file', tmp]);
const env = JSON.parse(await readFile(tmp, 'utf-8'));
for (const s of env.scannerEnvelope.scanners) {
for (const f of s.findings) {
assert.equal(f.userActionLanguage, undefined,
`${f.id}: --raw output-file must not carry userActionLanguage`);
}
}
} finally {
await unlink(tmp).catch(() => {});
}
});
});
});