BREAKING CHANGE: the {NNN} in CA-{SCANNER}-{NNN} identifies the check that
produced the finding. It used to be the finding's position in that scanner's
output for that run, which made it unstable across CONFIGURATIONS, not just
across releases as STATE framed it. Measured on two fixtures: "No custom
subagents" was CA-GAP-007 on minimal-project and CA-GAP-004 on healthy-project.
A user who fixed an unrelated earlier gap silently renumbered every later one,
so a .config-audit-ignore pin retargeted to a neighbouring finding with no
version change at all.
Second measured arm: README already documented the opposite scheme. It and the
scanner headers describe ~20 numbers as check codes (CA-SKL-003 = oversized
body, CA-PLH-015 = folder shadowing, CA-TOK-006 = schema deferral), and the
counter could only produce those in the all-fire case -- source-order positions
are 4, 3 and 8. The documentation described the scheme; the implementation was
what was wrong. Every published number is preserved by construction and pinned
exhaustively in tests/lib/finding-codes.test.mjs.
scanners/lib/finding-codes.mjs is the single authority. Every finding() call
passes a `code`; an undeclared or missing one THROWS. No counter fallback --
that would reproduce D1's findGapId -> 'unknown' silent degradation and let a
half-converted scanner ship IDs that look valid. findingCounter/resetCounter
are deleted outright, not left as no-ops. Retirement is now a mechanism:
RETIRED_CODES tombstones a withdrawn key so its number is never reissued,
seeded with GAP t3_8 -- the D1 removal that opened this chunk.
IDs are consequently NOT unique per finding: one check failing in three files
emits three findings sharing an ID. That inverts which consumer is correct, so
every f.id/findingId site was classified before the change. diff-engine and
most of fix-engine already keyed on scanner+title+file (drift was never lying);
fix-engine's verification did not, and keyed on the ID alone -- fixing one of
two sibling instances marked both fixed, and the untouched one, still present
in the re-scan, was reported as a REGRESSION. Red test first, then keyed on
(findingId, file), which both planFixes and applyFixes already carry.
plugin-health's crossIds Set was measured and is a clean negative: cross
findings are allFindings.slice(crossPluginStart) and codes 18/19 are emitted
only in that tail, so the partition holds by construction.
unknownSuppressions() reports a pin that names no declared check, in the
--output-file payload (ux-rules rule 2 -- a stderr-only warning is invisible to
the commands) and only when one exists, so a clean config is byte-identical.
That is what makes the break safe: a stale pin goes loud instead of dying quiet.
Frozen tests/snapshots/v5.0.0/ untouched on disk. IDs are masked out of that
comparison (mask-finding-ids.mjs) rather than re-derived -- re-deriving
positional IDs would assert the retired scheme against itself, and #58's
isGapEntry off-by-one is the measured example of that misfiring. The dead
re-derivation is removed from strip-retired-gap.mjs. default-output snapshots
re-approved after confirming the diff is IDs and nothing else.
Guards, each seen red against its own defect: a missing code (scanner errors
out mid-sweep), an orphan declaration, a resurrected retired key, and a
documented ID naming no check. The sweep asserts the union across all 16
scanners, never per scanner -- a per-scanner assertion goes green on a partial
conversion.
Fasit written before implementation: docs/mbug28-id-semantics-fasit.local.md,
including one correction made before running (CML has 12 checks over 13 call
sites -- the anchored and calibrated char-budget arms are one check, which a
repeated-title sweep found and my call-site count had missed).
Suite 1535 -> 1573, 0 failing.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MyqCQKK2ornJ1jFWwqx17E
271 lines
10 KiB
JavaScript
271 lines
10 KiB
JavaScript
/**
|
|
* RUL Scanner — Rules Validator
|
|
* Validates .claude/rules/ files: glob matching against real files, orphan detection, frontmatter.
|
|
* Finding IDs: CA-RUL-NNN
|
|
*/
|
|
|
|
import { readTextFile } from './lib/file-discovery.mjs';
|
|
import { finding, scannerResult } from './lib/output.mjs';
|
|
import { SEVERITY } from './lib/severity.mjs';
|
|
import { parseFrontmatter } from './lib/yaml-parser.mjs';
|
|
import { lineCount, truncate } from './lib/string-utils.mjs';
|
|
import { readdir, stat } from 'node:fs/promises';
|
|
import { join, resolve, relative, sep } from 'node:path';
|
|
|
|
const SCANNER = 'RUL';
|
|
|
|
/**
|
|
* Scan .claude/rules/ directories for issues.
|
|
* @param {string} targetPath
|
|
* @param {{ files: import('./lib/file-discovery.mjs').ConfigFile[] }} discovery
|
|
* @returns {Promise<object>}
|
|
*/
|
|
export async function scan(targetPath, discovery) {
|
|
const start = Date.now();
|
|
const ruleFiles = discovery.files.filter(f => f.type === 'rule');
|
|
const findings = [];
|
|
let filesScanned = 0;
|
|
|
|
if (ruleFiles.length === 0) {
|
|
return scannerResult(SCANNER, 'skipped', [], 0, Date.now() - start);
|
|
}
|
|
|
|
// Rule path patterns scope relative to the rule's OWN project root (the dir
|
|
// containing its .claude/), not the outer scan root. Resolve + cache per root.
|
|
const home = process.env.HOME || process.env.USERPROFILE || '';
|
|
const projectFilesByRoot = new Map();
|
|
async function projectFilesFor(root) {
|
|
if (!projectFilesByRoot.has(root)) {
|
|
projectFilesByRoot.set(root, await collectProjectFiles(root));
|
|
}
|
|
return projectFilesByRoot.get(root);
|
|
}
|
|
|
|
for (const file of ruleFiles) {
|
|
const content = await readTextFile(file.absPath);
|
|
if (!content) continue;
|
|
filesScanned++;
|
|
|
|
const { frontmatter, body, bodyStartLine } = parseFrontmatter(content);
|
|
const lines = lineCount(content);
|
|
|
|
// --- Frontmatter checks ---
|
|
if (!frontmatter) {
|
|
// Rules without frontmatter are "always on" — not necessarily wrong, just note it
|
|
if (lines > 5) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'no-frontmatter',
|
|
severity: SEVERITY.info,
|
|
title: 'Rule has no frontmatter (always active)',
|
|
description: `${file.relPath} has no YAML frontmatter. It will be loaded for ALL files. Add paths: frontmatter to scope it.`,
|
|
file: file.absPath,
|
|
recommendation: 'Add frontmatter with paths: to limit when this rule applies.',
|
|
}));
|
|
}
|
|
} else {
|
|
// Check for paths/globs frontmatter
|
|
const paths = frontmatter.paths || frontmatter.globs;
|
|
|
|
if (frontmatter.globs && !frontmatter.paths) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'globs-instead-of-paths',
|
|
severity: SEVERITY.low,
|
|
title: 'Rule uses "globs" instead of documented "paths"',
|
|
description: `${file.relPath} uses "globs:" for scoping. Claude Code's documentation specifies "paths:" as the rule-scoping field; "globs:" is not documented. Rename to "paths:" so the rule scopes as intended.`,
|
|
file: file.absPath,
|
|
evidence: `globs: ${JSON.stringify(frontmatter.globs)}`,
|
|
recommendation: 'Rename "globs:" to "paths:" — paths: is the documented field for path-scoped rules.',
|
|
autoFixable: true,
|
|
}));
|
|
}
|
|
|
|
if (paths) {
|
|
const patterns = Array.isArray(paths) ? paths : [paths];
|
|
|
|
// A rule scopes relative to its own project root (parent of its .claude/),
|
|
// not the scan root. User-global rules (root === HOME) match against the
|
|
// active project at runtime, so "matches no files here" is not meaningful.
|
|
const projectRoot = deriveProjectRoot(file.absPath) || targetPath;
|
|
const isUserGlobal = home && projectRoot === home;
|
|
|
|
if (!isUserGlobal) {
|
|
const projectFiles = await projectFilesFor(projectRoot);
|
|
|
|
for (const pattern of patterns) {
|
|
if (typeof pattern !== 'string') continue;
|
|
|
|
// Check if pattern matches any real files (relative to the rule's root)
|
|
const matchCount = countGlobMatches(pattern, projectFiles, projectRoot);
|
|
if (matchCount === 0) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'pattern-matches-nothing',
|
|
severity: SEVERITY.high,
|
|
title: 'Rule path pattern matches no files',
|
|
description: `${file.relPath}: pattern "${pattern}" matches 0 files. This rule will never activate.`,
|
|
file: file.absPath,
|
|
evidence: `paths: "${pattern}"`,
|
|
recommendation: 'Check the glob pattern. Common issues: wrong directory name, missing **, incorrect extension.',
|
|
autoFixable: false,
|
|
}));
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// --- Content quality checks ---
|
|
if (lines < 2) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'nearly-empty',
|
|
severity: SEVERITY.low,
|
|
title: 'Rule file is nearly empty',
|
|
description: `${file.relPath} has only ${lines} line(s).`,
|
|
file: file.absPath,
|
|
recommendation: 'Add meaningful content or remove the file.',
|
|
autoFixable: false,
|
|
}));
|
|
}
|
|
|
|
// Check for overly broad rules (huge files without path scoping)
|
|
if (!frontmatter?.paths && !frontmatter?.globs && lines > 50) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'large-unscoped',
|
|
severity: SEVERITY.medium,
|
|
title: 'Large unscoped rule file',
|
|
description: `${file.relPath} has ${lines} lines and no path scoping. It loads into context for every file interaction.`,
|
|
file: file.absPath,
|
|
evidence: `${lines} lines, no paths: frontmatter`,
|
|
recommendation: 'Add paths: frontmatter to scope this rule, or split into smaller path-specific rules.',
|
|
autoFixable: false,
|
|
}));
|
|
}
|
|
|
|
// A large PATH-SCOPED rule follows best practice, but path-scoped rules are
|
|
// NOT re-injected after a context compaction — they reload only when a
|
|
// matching file is read again (context-window.md). A big one carrying
|
|
// must-always-hold instructions can silently drop out mid-session.
|
|
if (frontmatter?.paths && lines > 50) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'large-scoped-lost-after-compaction',
|
|
severity: SEVERITY.low,
|
|
title: 'Large path-scoped rule is lost after compaction',
|
|
description: `${file.relPath} is path-scoped (${lines} lines). Path-scoped rules load only when a matching file is read, and after a context compaction they are not re-injected until a matching file is read again — so a large scoped rule carrying must-always-hold instructions can silently drop out mid-session.`,
|
|
file: file.absPath,
|
|
evidence: `${lines} lines, path-scoped`,
|
|
recommendation: 'If parts of this rule must always apply, move them to the project-root CLAUDE.md (re-injected after compaction). Keep path-scoped rules for context only needed when those files are open.',
|
|
autoFixable: false,
|
|
}));
|
|
}
|
|
|
|
// Check file extension
|
|
if (!file.absPath.endsWith('.md')) {
|
|
findings.push(finding({
|
|
scanner: SCANNER,
|
|
code: 'not-markdown',
|
|
severity: SEVERITY.medium,
|
|
title: 'Rule file is not .md',
|
|
description: `${file.relPath} is not a .md file. Only .md files are loaded from rules/.`,
|
|
file: file.absPath,
|
|
recommendation: 'Rename to .md extension.',
|
|
autoFixable: true,
|
|
}));
|
|
}
|
|
}
|
|
|
|
return scannerResult(SCANNER, 'ok', findings, filesScanned, Date.now() - start);
|
|
}
|
|
|
|
/**
|
|
* Collect project file paths for glob matching (limited depth).
|
|
* @param {string} targetPath
|
|
* @returns {Promise<string[]>}
|
|
*/
|
|
async function collectProjectFiles(targetPath, depth = 0) {
|
|
if (depth > 4) return [];
|
|
const SKIP = new Set(['node_modules', '.git', 'dist', 'build', 'coverage', '.next', '.nuxt', 'vendor']);
|
|
const files = [];
|
|
|
|
let entries;
|
|
try {
|
|
entries = await readdir(targetPath, { withFileTypes: true });
|
|
} catch {
|
|
return files;
|
|
}
|
|
|
|
for (const entry of entries) {
|
|
const fullPath = join(targetPath, entry.name);
|
|
if (entry.isFile()) {
|
|
files.push(fullPath);
|
|
} else if (entry.isDirectory() && !SKIP.has(entry.name) && !entry.name.startsWith('.')) {
|
|
const subFiles = await collectProjectFiles(fullPath, depth + 1);
|
|
files.push(...subFiles);
|
|
if (files.length > 5000) break; // Safety limit
|
|
}
|
|
}
|
|
|
|
return files;
|
|
}
|
|
|
|
/**
|
|
* Count how many files match a simplified glob pattern.
|
|
* Supports: *, **, specific extensions.
|
|
* @param {string} pattern
|
|
* @param {string[]} files
|
|
* @param {string} basePath
|
|
* @returns {number}
|
|
*/
|
|
/**
|
|
* Resolve the project root a rule scopes against: the directory containing the
|
|
* `.claude/` dir the rule lives under. `/a/b/.claude/rules/x.md` → `/a/b`.
|
|
* Returns null if the path has no `.claude` segment.
|
|
*/
|
|
function deriveProjectRoot(ruleAbsPath) {
|
|
const parts = ruleAbsPath.split(sep);
|
|
const idx = parts.lastIndexOf('.claude');
|
|
if (idx <= 0) return null;
|
|
return parts.slice(0, idx).join(sep);
|
|
}
|
|
|
|
function countGlobMatches(pattern, files, basePath) {
|
|
try {
|
|
const regex = globToRegex(pattern);
|
|
let count = 0;
|
|
for (const file of files) {
|
|
const rel = relative(basePath, file);
|
|
if (regex.test(rel)) count++;
|
|
}
|
|
return count;
|
|
} catch {
|
|
return -1; // Pattern parsing error — don't report as orphan
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Convert a simple glob pattern to a regex.
|
|
* Handles ** matching zero or more path segments.
|
|
* @param {string} pattern
|
|
* @returns {RegExp}
|
|
*/
|
|
function globToRegex(pattern) {
|
|
let regex = pattern
|
|
.replace(/\./g, '\\.')
|
|
.replace(/\/\*\*\//g, '{{GLOBSTAR_SLASH}}')
|
|
.replace(/\*\*/g, '{{GLOBSTAR}}')
|
|
.replace(/\*/g, '[^/]*')
|
|
.replace(/\?/g, '[^/]') // must run BEFORE placeholder restore — '(?:' would corrupt
|
|
.replace(/\{\{GLOBSTAR_SLASH\}\}/g, '(?:/.+/|/)') // **/ matches 0+ intermediate dirs
|
|
.replace(/\{\{GLOBSTAR\}\}/g, '.*');
|
|
|
|
// Handle leading patterns
|
|
if (!regex.startsWith('.*') && !regex.startsWith('/')) {
|
|
regex = '(?:^|/)' + regex;
|
|
}
|
|
|
|
return new RegExp(regex);
|
|
}
|