feat(skl,cml): --context-window calibration, advisory when unknown (v5.11 B8) [skip-docs]
SKL-002 (skill-listing budget) and CML char-budget now calibrate to a
resolved context window instead of always anchoring at 200k:
- resolveContextWindow(): --context-window <n> calibrates; 'auto' keeps the
conservative 200k anchor but marks advisory (model→window probing deferred
to B8b); no flag → 200k anchor, byte-identical to pre-B8 default.
- scaleForWindow(): linear off the 200k anchor (identity at the anchor).
- SKL + CML each keep an untouched default branch (window===200k && !advisory)
for byte-stability and a calibrated branch; advisory downgrades the budget
finding from a breach (low/medium) to info.
- Flag wired through scan-orchestrator + posture; runAllScanners resolves once
and threads { contextWindow } to scanners (others ignore the 3rd arg).
- CPS intentionally excluded: it has no window-anchored budget (fixed
150-line volatility heuristic), so there is nothing to calibrate.
15 new tests; e2e CLI verified (1M suppresses SKL-002, auto → info, default
unchanged); full suite 1279 green; snapshots byte-stable.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
27988801be
commit
2082b7d112
10 changed files with 364 additions and 45 deletions
|
|
@ -77,23 +77,26 @@ export const BODY_CALIBRATION_NOTE =
|
|||
* flags it — so the aggregate does not double-count it).
|
||||
*
|
||||
* @param {number[]} descLengths - one entry per active skill (description char count)
|
||||
* @param {number} [budgetTokens=AGGREGATE_BUDGET_TOKENS] - the listing budget to
|
||||
* measure against. Defaults to the 200k-anchored 4,000 tok; B8 passes a
|
||||
* window-calibrated budget. Defaulting keeps existing callers byte-stable.
|
||||
* @returns {BudgetAssessment}
|
||||
*/
|
||||
export function assessSkillListingBudget(descLengths) {
|
||||
export function assessSkillListingBudget(descLengths, budgetTokens = AGGREGATE_BUDGET_TOKENS) {
|
||||
let aggregateChars = 0;
|
||||
for (const len of descLengths) {
|
||||
const safe = (typeof len === 'number' && Number.isFinite(len) && len > 0) ? len : 0;
|
||||
aggregateChars += Math.min(safe, DESCRIPTION_CAP);
|
||||
}
|
||||
const aggregateTokens = estimateTokens(aggregateChars, 'markdown');
|
||||
const overBudget = aggregateTokens > AGGREGATE_BUDGET_TOKENS;
|
||||
const overBudget = aggregateTokens > budgetTokens;
|
||||
return {
|
||||
scanned: descLengths.length,
|
||||
aggregateChars,
|
||||
aggregateTokens,
|
||||
budgetTokens: AGGREGATE_BUDGET_TOKENS,
|
||||
budgetTokens,
|
||||
overBudget,
|
||||
overBy: overBudget ? aggregateTokens - AGGREGATE_BUDGET_TOKENS : 0,
|
||||
overBy: overBudget ? aggregateTokens - budgetTokens : 0,
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -115,9 +118,11 @@ export function assessSkillListingBudget(descLengths) {
|
|||
* enumerateSkills). Callers that run under test MUST override HOME (see the
|
||||
* hermetic-home helper / runScannerWithHome pattern).
|
||||
*
|
||||
* @param {number} [budgetTokens=AGGREGATE_BUDGET_TOKENS] - listing budget for the
|
||||
* aggregate assessment (B8 window-calibration); defaults keep callers byte-stable.
|
||||
* @returns {Promise<{ skills: ActiveSkillEntry[], aggregate: BudgetAssessment }>}
|
||||
*/
|
||||
export async function measureActiveSkillListing() {
|
||||
export async function measureActiveSkillListing(budgetTokens = AGGREGATE_BUDGET_TOKENS) {
|
||||
const plugins = await enumeratePlugins();
|
||||
const allSkills = await enumerateSkills(plugins);
|
||||
|
||||
|
|
@ -143,7 +148,7 @@ export async function measureActiveSkillListing() {
|
|||
});
|
||||
}
|
||||
|
||||
const aggregate = assessSkillListingBudget(skills.map((s) => s.descLength));
|
||||
const aggregate = assessSkillListingBudget(skills.map((s) => s.descLength), budgetTokens);
|
||||
return { skills, aggregate };
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue