BREAKING CHANGE: the {NNN} in CA-{SCANNER}-{NNN} identifies the check that
produced the finding. It used to be the finding's position in that scanner's
output for that run, which made it unstable across CONFIGURATIONS, not just
across releases as STATE framed it. Measured on two fixtures: "No custom
subagents" was CA-GAP-007 on minimal-project and CA-GAP-004 on healthy-project.
A user who fixed an unrelated earlier gap silently renumbered every later one,
so a .config-audit-ignore pin retargeted to a neighbouring finding with no
version change at all.
Second measured arm: README already documented the opposite scheme. It and the
scanner headers describe ~20 numbers as check codes (CA-SKL-003 = oversized
body, CA-PLH-015 = folder shadowing, CA-TOK-006 = schema deferral), and the
counter could only produce those in the all-fire case -- source-order positions
are 4, 3 and 8. The documentation described the scheme; the implementation was
what was wrong. Every published number is preserved by construction and pinned
exhaustively in tests/lib/finding-codes.test.mjs.
scanners/lib/finding-codes.mjs is the single authority. Every finding() call
passes a `code`; an undeclared or missing one THROWS. No counter fallback --
that would reproduce D1's findGapId -> 'unknown' silent degradation and let a
half-converted scanner ship IDs that look valid. findingCounter/resetCounter
are deleted outright, not left as no-ops. Retirement is now a mechanism:
RETIRED_CODES tombstones a withdrawn key so its number is never reissued,
seeded with GAP t3_8 -- the D1 removal that opened this chunk.
IDs are consequently NOT unique per finding: one check failing in three files
emits three findings sharing an ID. That inverts which consumer is correct, so
every f.id/findingId site was classified before the change. diff-engine and
most of fix-engine already keyed on scanner+title+file (drift was never lying);
fix-engine's verification did not, and keyed on the ID alone -- fixing one of
two sibling instances marked both fixed, and the untouched one, still present
in the re-scan, was reported as a REGRESSION. Red test first, then keyed on
(findingId, file), which both planFixes and applyFixes already carry.
plugin-health's crossIds Set was measured and is a clean negative: cross
findings are allFindings.slice(crossPluginStart) and codes 18/19 are emitted
only in that tail, so the partition holds by construction.
unknownSuppressions() reports a pin that names no declared check, in the
--output-file payload (ux-rules rule 2 -- a stderr-only warning is invisible to
the commands) and only when one exists, so a clean config is byte-identical.
That is what makes the break safe: a stale pin goes loud instead of dying quiet.
Frozen tests/snapshots/v5.0.0/ untouched on disk. IDs are masked out of that
comparison (mask-finding-ids.mjs) rather than re-derived -- re-deriving
positional IDs would assert the retired scheme against itself, and #58's
isGapEntry off-by-one is the measured example of that misfiring. The dead
re-derivation is removed from strip-retired-gap.mjs. default-output snapshots
re-approved after confirming the diff is IDs and nothing else.
Guards, each seen red against its own defect: a missing code (scanner errors
out mid-sweep), an orphan declaration, a resurrected retired key, and a
documented ID naming no check. The sweep asserts the union across all 16
scanners, never per scanner -- a per-scanner assertion goes green on a partial
conversion.
Fasit written before implementation: docs/mbug28-id-semantics-fasit.local.md,
including one correction made before running (CML has 12 checks over 13 call
sites -- the anchored and calibrated char-budget arms are one check, which a
repeated-title sweep found and my call-site count had missed).
Suite 1535 -> 1573, 0 failing.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MyqCQKK2ornJ1jFWwqx17E
340 lines
13 KiB
JavaScript
340 lines
13 KiB
JavaScript
import { describe, it } from 'node:test';
|
|
import assert from 'node:assert/strict';
|
|
import { resolve, dirname } from 'node:path';
|
|
import { fileURLToPath } from 'node:url';
|
|
import { execFile } from 'node:child_process';
|
|
import { promisify } from 'node:util';
|
|
import { readFile, writeFile, unlink, mkdir, access } from 'node:fs/promises';
|
|
import { hermeticEnv, HERMETIC_HOME } from '../helpers/hermetic-home.mjs';
|
|
import { stripHotspotLoadPattern } from '../helpers/strip-hotspot-load-pattern.mjs';
|
|
import { stripRetiredGap } from '../helpers/strip-retired-gap.mjs';
|
|
import { maskFindingIds } from '../helpers/mask-finding-ids.mjs';
|
|
|
|
const exec = promisify(execFile);
|
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
const REPO = resolve(__dirname, '../..');
|
|
const FIXTURE = resolve(REPO, 'tests/fixtures/marketplace-medium');
|
|
const BROKEN_PLUGIN = resolve(REPO, 'tests/fixtures/broken-plugin');
|
|
|
|
const BASELINE_DIR = resolve(HERMETIC_HOME, '.config-audit/baselines');
|
|
const DEFAULT_BASELINE = resolve(BASELINE_DIR, 'default.json');
|
|
|
|
/**
|
|
* Run a CLI subprocess and return stdout/stderr regardless of exit code
|
|
* (some CLIs exit non-zero on findings — we still need their output).
|
|
*/
|
|
async function runCli(cliPath, args, env = {}) {
|
|
try {
|
|
const { stdout, stderr } = await exec('node', [cliPath, ...args], {
|
|
timeout: 60000,
|
|
cwd: REPO,
|
|
env: hermeticEnv(env),
|
|
maxBuffer: 10 * 1024 * 1024,
|
|
});
|
|
return { stdout: stdout || '', stderr: stderr || '', code: 0 };
|
|
} catch (err) {
|
|
return {
|
|
stdout: err.stdout || '',
|
|
stderr: err.stderr || '',
|
|
code: err.code ?? 1,
|
|
};
|
|
}
|
|
}
|
|
|
|
/** Strip time-varying duration_ms / Xms occurrences for snapshot comparison. */
|
|
function normalizeTokenHotspotsPayload(p) {
|
|
const out = JSON.parse(JSON.stringify(p));
|
|
out.duration_ms = 0;
|
|
return maskFindingIds(stripRetiredGap(stripHotspotLoadPattern(out)));
|
|
}
|
|
|
|
function normalizeManifestOutput(o) {
|
|
const out = JSON.parse(JSON.stringify(o));
|
|
if (out.meta) {
|
|
out.meta.repoPath = '<TARGET>';
|
|
out.meta.generatedAt = '<TIMESTAMP>';
|
|
out.meta.durationMs = 0;
|
|
}
|
|
return out;
|
|
}
|
|
|
|
function normalizeWhatsActiveOutput(o) {
|
|
const out = JSON.parse(JSON.stringify(o));
|
|
if (out.meta) {
|
|
out.meta.repoPath = '<TARGET>';
|
|
out.meta.generatedAt = '<TIMESTAMP>';
|
|
out.meta.durationMs = 0;
|
|
if (out.meta.gitRoot) out.meta.gitRoot = '<GITROOT>';
|
|
if (out.meta.projectKey) out.meta.projectKey = '<PROJECTKEY>';
|
|
}
|
|
return out;
|
|
}
|
|
|
|
function normalizePluginHealthOutput(o) {
|
|
const out = JSON.parse(JSON.stringify(o));
|
|
out.duration_ms = 0;
|
|
return out;
|
|
}
|
|
|
|
function normalizeDriftOutput(o) {
|
|
// Drift result has no time fields; just round-trip through JSON.
|
|
return maskFindingIds(stripRetiredGap(JSON.parse(JSON.stringify(o))));
|
|
}
|
|
|
|
// ============================================================================
|
|
// token-hotspots-cli
|
|
// ============================================================================
|
|
describe('token-hotspots-cli humanizer (Step 7)', () => {
|
|
const CLI = resolve(REPO, 'scanners/token-hotspots-cli.mjs');
|
|
const SNAPSHOT = resolve(REPO, 'tests/snapshots/v5.0.0/token-hotspots.json');
|
|
|
|
it('--json: payload.findings byte-equal v5.0.0 snapshot', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const actual = JSON.parse(stdout);
|
|
const expected = JSON.parse(await readFile(SNAPSHOT, 'utf-8'));
|
|
assert.deepStrictEqual(
|
|
normalizeTokenHotspotsPayload(actual),
|
|
normalizeTokenHotspotsPayload(expected),
|
|
);
|
|
});
|
|
|
|
it('--raw: payload.findings byte-equal v5.0.0 snapshot', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
const actual = JSON.parse(stdout);
|
|
const expected = JSON.parse(await readFile(SNAPSHOT, 'utf-8'));
|
|
assert.deepStrictEqual(
|
|
normalizeTokenHotspotsPayload(actual),
|
|
normalizeTokenHotspotsPayload(expected),
|
|
);
|
|
});
|
|
|
|
it('default: payload.findings include humanizer fields when findings exist', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE]);
|
|
const actual = JSON.parse(stdout);
|
|
if (actual.findings.length === 0) return;
|
|
for (const f of actual.findings) {
|
|
assert.equal(typeof f.userImpactCategory, 'string',
|
|
`${f.id}: default mode must add userImpactCategory`);
|
|
assert.equal(typeof f.userActionLanguage, 'string',
|
|
`${f.id}: default mode must add userActionLanguage`);
|
|
assert.equal(typeof f.relevanceContext, 'string',
|
|
`${f.id}: default mode must add relevanceContext`);
|
|
}
|
|
});
|
|
|
|
it('--json: payload.findings do NOT carry humanizer fields', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const actual = JSON.parse(stdout);
|
|
for (const f of actual.findings) {
|
|
assert.equal(f.userImpactCategory, undefined,
|
|
`${f.id}: --json must not add userImpactCategory`);
|
|
}
|
|
});
|
|
});
|
|
|
|
// ============================================================================
|
|
// plugin-health-scanner
|
|
//
|
|
// NOTE: plugin-health scans the plugin root (not the fixture path), so its
|
|
// findings reflect the current marketplace state — snapshot frozen at Wave 0
|
|
// no longer matches as new plugins are added. We verify mode-equivalence
|
|
// (--json == --raw) instead.
|
|
// ============================================================================
|
|
describe('plugin-health-scanner humanizer (Step 7)', () => {
|
|
const CLI = resolve(REPO, 'scanners/plugin-health-scanner.mjs');
|
|
|
|
it('--json and --raw produce byte-identical stdout (both bypass humanizer)', async () => {
|
|
const { stdout: jsonOut } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const { stdout: rawOut } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
assert.deepStrictEqual(
|
|
normalizePluginHealthOutput(JSON.parse(jsonOut)),
|
|
normalizePluginHealthOutput(JSON.parse(rawOut)),
|
|
);
|
|
});
|
|
|
|
it('--json output preserves v5.0.0 finding shape (no humanizer fields)', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const actual = JSON.parse(stdout);
|
|
for (const f of actual.findings || []) {
|
|
assert.equal(f.userImpactCategory, undefined,
|
|
`${f.id}: --json must not add userImpactCategory`);
|
|
}
|
|
});
|
|
|
|
it('default mode renders to stderr (humanized when findings exist)', async () => {
|
|
const { stderr: defaultStderr } = await runCli(CLI, [BROKEN_PLUGIN]);
|
|
const { stderr: rawStderr } = await runCli(CLI, [BROKEN_PLUGIN, '--raw']);
|
|
// --raw suppresses prose stderr (machine mode); default emits humanized prose.
|
|
// Just verify both run without crash; humanization assertion is best-effort
|
|
// because broken-plugin may produce no PLH-translated findings.
|
|
assert.ok(typeof defaultStderr === 'string');
|
|
assert.ok(typeof rawStderr === 'string');
|
|
});
|
|
});
|
|
|
|
// ============================================================================
|
|
// drift-cli
|
|
// ============================================================================
|
|
describe('drift-cli humanizer (Step 7)', () => {
|
|
const CLI = resolve(REPO, 'scanners/drift-cli.mjs');
|
|
const SNAPSHOT = resolve(REPO, 'tests/snapshots/v5.0.0/drift.json');
|
|
|
|
async function ensureBaseline() {
|
|
try {
|
|
await access(DEFAULT_BASELINE);
|
|
return true;
|
|
} catch {
|
|
// Try to save one
|
|
try {
|
|
await mkdir(BASELINE_DIR, { recursive: true });
|
|
await runCli(CLI, [FIXTURE, '--save']);
|
|
await access(DEFAULT_BASELINE);
|
|
return true;
|
|
} catch {
|
|
return false;
|
|
}
|
|
}
|
|
}
|
|
|
|
it('--json: diff byte-equal v5.0.0 snapshot', async () => {
|
|
const ok = await ensureBaseline();
|
|
if (!ok) {
|
|
// SKIP — baseline cannot be created in this environment.
|
|
return;
|
|
}
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const actual = JSON.parse(stdout);
|
|
const expected = JSON.parse(await readFile(SNAPSHOT, 'utf-8'));
|
|
assert.deepStrictEqual(
|
|
normalizeDriftOutput(actual),
|
|
normalizeDriftOutput(expected),
|
|
);
|
|
});
|
|
|
|
it('--raw: diff byte-equal v5.0.0 snapshot', async () => {
|
|
const ok = await ensureBaseline();
|
|
if (!ok) return;
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
const actual = JSON.parse(stdout);
|
|
const expected = JSON.parse(await readFile(SNAPSHOT, 'utf-8'));
|
|
assert.deepStrictEqual(
|
|
normalizeDriftOutput(actual),
|
|
normalizeDriftOutput(expected),
|
|
);
|
|
});
|
|
|
|
it('default: stderr report differs from --raw stderr when findings exist', async () => {
|
|
const ok = await ensureBaseline();
|
|
if (!ok) return;
|
|
const { stderr: defaultStderr } = await runCli(CLI, [FIXTURE]);
|
|
const { stderr: rawStderr } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
// If there are findings whose titles get humanized, default stderr differs from raw.
|
|
// If no humanizable titles in this fixture, both can match — just verify no crash.
|
|
assert.ok(typeof defaultStderr === 'string');
|
|
assert.ok(typeof rawStderr === 'string');
|
|
});
|
|
});
|
|
|
|
// ============================================================================
|
|
// manifest
|
|
//
|
|
// NOTE: manifest scans the active config cascade (env-dependent), so the
|
|
// frozen v5.0.0 snapshot drifts as the marketplace changes. We verify
|
|
// --json == --raw == default (no-op for inventory) instead.
|
|
// ============================================================================
|
|
describe('manifest humanizer (Step 7) — no-op for --raw', () => {
|
|
const CLI = resolve(REPO, 'scanners/manifest.mjs');
|
|
|
|
it('--json and --raw produce byte-identical output', async () => {
|
|
const { stdout: jsonOut } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const { stdout: rawOut } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
assert.deepStrictEqual(
|
|
normalizeManifestOutput(JSON.parse(jsonOut)),
|
|
normalizeManifestOutput(JSON.parse(rawOut)),
|
|
);
|
|
});
|
|
|
|
it('default and --raw produce structurally identical output (inventory CLI)', async () => {
|
|
const { stdout: defaultOut } = await runCli(CLI, [FIXTURE]);
|
|
const { stdout: rawOut } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
assert.deepStrictEqual(
|
|
normalizeManifestOutput(JSON.parse(defaultOut)),
|
|
normalizeManifestOutput(JSON.parse(rawOut)),
|
|
);
|
|
});
|
|
|
|
it('preserves v5.0.0 envelope shape', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const out = JSON.parse(stdout);
|
|
assert.ok(out.meta);
|
|
assert.ok(Array.isArray(out.sources));
|
|
assert.equal(typeof out.total, 'number');
|
|
});
|
|
});
|
|
|
|
// ============================================================================
|
|
// whats-active
|
|
//
|
|
// NOTE: whats-active scans the active config (env-dependent). Frozen snapshot
|
|
// drifts; we verify mode-equivalence instead.
|
|
// ============================================================================
|
|
describe('whats-active humanizer (Step 7) — no-op for --raw', () => {
|
|
const CLI = resolve(REPO, 'scanners/whats-active.mjs');
|
|
|
|
it('--json and --raw produce byte-identical output', async () => {
|
|
const { stdout: jsonOut } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const { stdout: rawOut } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
assert.deepStrictEqual(
|
|
normalizeWhatsActiveOutput(JSON.parse(jsonOut)),
|
|
normalizeWhatsActiveOutput(JSON.parse(rawOut)),
|
|
);
|
|
});
|
|
|
|
it('default and --raw produce structurally identical output (inventory CLI)', async () => {
|
|
const { stdout: defaultOut } = await runCli(CLI, [FIXTURE]);
|
|
const { stdout: rawOut } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
assert.deepStrictEqual(
|
|
normalizeWhatsActiveOutput(JSON.parse(defaultOut)),
|
|
normalizeWhatsActiveOutput(JSON.parse(rawOut)),
|
|
);
|
|
});
|
|
|
|
it('preserves v5.0.0 envelope shape', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const out = JSON.parse(stdout);
|
|
assert.ok(out.meta);
|
|
assert.ok(out.claudeMd);
|
|
assert.ok(Array.isArray(out.plugins));
|
|
assert.ok(Array.isArray(out.skills));
|
|
});
|
|
});
|
|
|
|
// ============================================================================
|
|
// fix-cli
|
|
// ============================================================================
|
|
describe('fix-cli humanizer (Step 7)', () => {
|
|
const CLI = resolve(REPO, 'scanners/fix-cli.mjs');
|
|
const SNAPSHOT = resolve(REPO, 'tests/snapshots/v5.0.0/fix-cli.json');
|
|
|
|
it('--json: stdout JSON byte-equal v5.0.0 snapshot', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--json']);
|
|
const actual = JSON.parse(stdout);
|
|
const expected = JSON.parse(await readFile(SNAPSHOT, 'utf-8'));
|
|
assert.deepStrictEqual(maskFindingIds(stripRetiredGap(actual)), maskFindingIds(stripRetiredGap(expected)));
|
|
});
|
|
|
|
it('--raw: stdout JSON byte-equal v5.0.0 snapshot', async () => {
|
|
const { stdout } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
const actual = JSON.parse(stdout);
|
|
const expected = JSON.parse(await readFile(SNAPSHOT, 'utf-8'));
|
|
assert.deepStrictEqual(maskFindingIds(stripRetiredGap(actual)), maskFindingIds(stripRetiredGap(expected)));
|
|
});
|
|
|
|
it('default mode stderr differs from --raw stderr when findings have humanizer translations', async () => {
|
|
const { stderr: defaultStderr } = await runCli(CLI, [FIXTURE]);
|
|
const { stderr: rawStderr } = await runCli(CLI, [FIXTURE, '--raw']);
|
|
// 20 manual findings in fixture; many have GAP translations → stderr differs.
|
|
assert.notEqual(defaultStderr, rawStderr,
|
|
'fix-cli default stderr must differ from --raw stderr when humanizer translates titles');
|
|
});
|
|
});
|