fix(acr): token estimator discounts block-level HTML comments (M-BUG-6)
CLAUDE.md token estimates counted block-level <!-- --> HTML comments toward always-loaded tokens, but CC strips them before injection (preserved only inside code fences, per code.claude.com/docs/en/memory). Fix: new stripInjectedHtmlComments + effectiveMemoryBytes in active-config-reader; the CML cascade (walkClaudeMdCascade) and token-hotspots now size CLAUDE.md from effective (stripped) bytes, while raw byte figures stay honest. Block-level only — inline comments retained (conservative, verified scope). Suite 1329/0 (+13). Frozen v5.0.0 snapshots untouched (no fixture has <!--), no re-seed. Dogfood ~/.claude CLAUDE.md ~3386->3301 tok (~85 tok discount, matches worklist prediction).
This commit is contained in:
parent
dd9db60fc9
commit
7e94910566
4 changed files with 234 additions and 6 deletions
|
|
@ -22,12 +22,12 @@
|
|||
*/
|
||||
|
||||
import { resolve, dirname, isAbsolute } from 'node:path';
|
||||
import { stat } from 'node:fs/promises';
|
||||
import { stat, readFile } from 'node:fs/promises';
|
||||
import { readTextFile } from './lib/file-discovery.mjs';
|
||||
import { finding, scannerResult } from './lib/output.mjs';
|
||||
import { SEVERITY } from './lib/severity.mjs';
|
||||
import { findImports, parseJson, parseFrontmatter } from './lib/yaml-parser.mjs';
|
||||
import { estimateTokens, readActiveConfig, deriveLoadPattern } from './lib/active-config-reader.mjs';
|
||||
import { estimateTokens, effectiveMemoryBytes, readActiveConfig, deriveLoadPattern } from './lib/active-config-reader.mjs';
|
||||
import {
|
||||
assessMcpDeferralForRepo,
|
||||
severityForForcedSchemas,
|
||||
|
|
@ -261,6 +261,26 @@ function detectRedundantPermissions(settings) {
|
|||
return issues;
|
||||
}
|
||||
|
||||
/**
|
||||
* Byte count to feed the token estimator for a discovered file. CLAUDE.md /
|
||||
* memory files are sized from their *effective* (injected) content — CC strips
|
||||
* block-level HTML comments before injection — so a raw byte read over-counts
|
||||
* them. Every other source uses the raw on-disk size. (M-BUG-6)
|
||||
*
|
||||
* @param {{type:string, absPath?:string, size:number}} f
|
||||
* @returns {Promise<number>}
|
||||
*/
|
||||
async function tokenBytesFor(f) {
|
||||
if (f.type === 'claude-md' && f.absPath) {
|
||||
try {
|
||||
return effectiveMemoryBytes(await readFile(f.absPath, 'utf-8'));
|
||||
} catch {
|
||||
return f.size;
|
||||
}
|
||||
}
|
||||
return f.size;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the ranked hotspots array.
|
||||
*
|
||||
|
|
@ -272,7 +292,7 @@ async function buildHotspots(discovery, targetPath, activeConfig) {
|
|||
const ranked = [];
|
||||
for (const f of discovery.files) {
|
||||
const kind = tokenKind(f.type);
|
||||
const tokens = estimateTokens(f.size, kind);
|
||||
const tokens = estimateTokens(await tokenBytesFor(f), kind);
|
||||
if (tokens <= 0) continue;
|
||||
ranked.push({
|
||||
absPath: f.absPath,
|
||||
|
|
@ -651,7 +671,7 @@ export async function scan(targetPath, discovery) {
|
|||
// ── Total estimated tokens (sum of every discovered source + activeConfig MCP) ──
|
||||
let totalTokens = 0;
|
||||
for (const f of discovery.files) {
|
||||
totalTokens += estimateTokens(f.size, tokenKind(f.type));
|
||||
totalTokens += estimateTokens(await tokenBytesFor(f), tokenKind(f.type));
|
||||
}
|
||||
if (activeConfig && Array.isArray(activeConfig.mcpServers)) {
|
||||
for (const m of activeConfig.mcpServers) {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue