feat(ms-ai-architect): Spor D steg 2 — score-skill CLI (--json/--gate/--skill) (TDD)
Byggekloss for både SessionStart-surfacing og lifecycle-gate (D3). - eval.mjs: ekstraher buildReport() (eksportert, testbar) ut av main() — samler hele korpus-rapporten (deterministisk + K10 + merget judge-cache). - lib/skill-score.mjs: formatScoreReport() (ren) — sortert forbedringsrapport med score, status, floored/provisional/ujudget-tags og ⚑gulv-markering. - score-skill.mjs (NY CLI): buildReport → scoreReport → render/json/gate. --skill scorer hele korpuset (K10 trenger alle søsken) men rapporterer/gater kun den valgte → inkrementell enkelt-skill-scoring for D3 lifecycle-gate. - docs/development.md: ny «Skill-kvalitetsscore (Spor D)»-seksjon. Live baseline: advisor 91, governance/security 96 (OK); engineering + infrastructure 89 (floored av K10-gulv) UNDER mål. Gate exit=1. Tester: kb-eval 134→142 (+8). validate 239, kb-update 316 uendret grønt. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
e87289d929
commit
c1a09062d4
5 changed files with 249 additions and 7 deletions
|
|
@ -93,6 +93,21 @@ Et parallelt, uavhengig spor fra sitemap-discoveren: oppdage at Microsoft har pu
|
|||
|
||||
Slug→skill-mapping er config i `domain-taxonomy.json` (`course_products`), ikke hardkoding. `last_full_enum`-kadens (≥30 d) styrer når `removed` beregnes (kun full-enumerering — aldri inkrementelt).
|
||||
|
||||
### Skill-kvalitetsscore (Spor D — objektiv 0–100-score per skill)
|
||||
|
||||
Måler skill-kvalitet deterministisk mot Anthropics «Skill authoring best practices». Rubrikken K1–K10 (`eval.mjs`) + de fem ekstra deterministiske sjekkene N1–N5 aggregeres til én 0–100-score per skill via **vektet delpoeng med hardt gulv** på de bærende kriteriene (K1 trigger-presisjon, K10 søsken-overlapp): feiler ett gulv-kriterium kappes scoren under målet (89), uansett øvrig form. Mål: **alle 5 skills ≥90 %**. Full spec: [`skill-quality-scoring-plan.md`](skill-quality-scoring-plan.md).
|
||||
|
||||
```bash
|
||||
node scripts/kb-eval/score-skill.mjs # human-rapport: score + sortert forbedringsliste per skill
|
||||
node scripts/kb-eval/score-skill.mjs --json # maskin-output (scoreReport: target/scored/below)
|
||||
node scripts/kb-eval/score-skill.mjs --gate 90 # non-zero exit hvis noen skill < 90
|
||||
node scripts/kb-eval/score-skill.mjs --skill ms-ai-advisor # scope til én skill (inkrementelt)
|
||||
```
|
||||
|
||||
Arkitektur: `eval.mjs buildReport()` (disk: deterministisk + K10 + merget operatør-gated judge-cache) → `lib/skill-score.mjs` (REN: `scoreSkill`/`scoreReport`/`formatScoreReport`, ingen disk). N1 name-validitet · N2 description ≤1024 · N3 refs én nivå dypt · N4 TOC i ref-filer >100 linjer · N5 forward-slash-stier. Judge-kriterier (K1/K4/K7/K8/K9) degraderer pent: mangler judge → ekskluderes fra teller+nevner, scoren merkes `provisional` (K1-gulv kan ikke håndheves), K10-gulvet (deterministisk) gjelder alltid.
|
||||
|
||||
**Korpus-invariant (roadmap):** Hele korpuset skal holde ≥90 % til enhver tid. Ved skill-endring (create/update/merge/split via `apply-skill-op.mjs`) re-scores **kun** den endrede skillen inkrementelt (men hele korpuset evalueres siden K10 trenger alle søsken-descriptions), og `--gate 90` håndhever terskelen via operatør-gate.
|
||||
|
||||
### E2E-regresjonstester
|
||||
```bash
|
||||
# Kjør alle E2E-suiter
|
||||
|
|
|
|||
|
|
@ -334,11 +334,12 @@ export function attachSiblingOverlap(skills, promptSet, { threshold = K10_OVERLA
|
|||
return skills;
|
||||
}
|
||||
|
||||
function main() {
|
||||
const args = process.argv.slice(2);
|
||||
const jsonOut = args.includes('--json');
|
||||
const doWrite = args.includes('--write');
|
||||
|
||||
/**
|
||||
* Assemble the full corpus report: per-skill deterministic eval (incl. N1-N5) +
|
||||
* cross-skill K10 sibling-overlap + merged operator-gated judge cache. This is
|
||||
* the object the score-skill CLI consumes via scoreReport(). Reads disk.
|
||||
*/
|
||||
export function buildReport() {
|
||||
const skillNames = readdirSync(SKILLS_DIR, { withFileTypes: true })
|
||||
.filter((e) => e.isDirectory() && existsSync(join(SKILLS_DIR, e.name, 'SKILL.md')))
|
||||
.map((e) => e.name)
|
||||
|
|
@ -360,11 +361,20 @@ function main() {
|
|||
for (const s of skills) if (jr[s.name]) s.judge = jr[s.name];
|
||||
}
|
||||
|
||||
const report = {
|
||||
return {
|
||||
rubric: 'K1-K10',
|
||||
note: 'Deterministic: K2,K3,K5,K6,refCountConsistency,K10(siblingScopeOverlap). LLM-judge (operator-gated): K1,K4,K7,K8,K9.',
|
||||
note: 'Deterministic: K2,K3,K5,K6,refCountConsistency,K10(siblingScopeOverlap),N1-N5. LLM-judge (operator-gated): K1,K4,K7,K8,K9.',
|
||||
skills,
|
||||
};
|
||||
}
|
||||
|
||||
function main() {
|
||||
const args = process.argv.slice(2);
|
||||
const jsonOut = args.includes('--json');
|
||||
const doWrite = args.includes('--write');
|
||||
|
||||
const report = buildReport();
|
||||
const skills = report.skills;
|
||||
|
||||
if (doWrite) {
|
||||
mkdirSync(OUT_DIR, { recursive: true });
|
||||
|
|
|
|||
|
|
@ -249,3 +249,35 @@ export function scoreReport(report, opts = {}) {
|
|||
const below = scored.filter((s) => !s.meetsTarget);
|
||||
return { target, scored, below };
|
||||
}
|
||||
|
||||
/**
|
||||
* Render a human-readable quality report (pure). Each skill shows its score and
|
||||
* status; below-target skills list their sorted improvements (biggest weighted
|
||||
* loss first). Returns a multi-line string.
|
||||
* @param {{target: number, scored: object[]}} result scoreReport() output
|
||||
*/
|
||||
export function formatScoreReport(result) {
|
||||
const target = result?.target ?? TARGET;
|
||||
const scored = result?.scored || [];
|
||||
const below = scored.filter((s) => !s.meetsTarget);
|
||||
const lines = [];
|
||||
lines.push(`Skill-kvalitet — ${scored.length} skills, mål ≥${target} % (${below.length} under mål)`);
|
||||
lines.push('');
|
||||
for (const s of scored) {
|
||||
const tags = [];
|
||||
if (s.floored) tags.push('floored');
|
||||
if (s.provisional) tags.push('provisional');
|
||||
if (!s.judged) tags.push('ujudget');
|
||||
const status = s.meetsTarget ? 'OK' : 'UNDER MÅL';
|
||||
const tagStr = tags.length ? ` [${tags.join(', ')}]` : '';
|
||||
lines.push(`■ ${s.name} ${s.score}/100 ${status}${tagStr}`);
|
||||
if (!s.meetsTarget) {
|
||||
for (const imp of s.improvements) {
|
||||
const floor = imp.floor ? ' ⚑gulv' : '';
|
||||
lines.push(` ↳ ${imp.key} ${imp.label}${floor} (${imp.detail}) [-${imp.pointsLost.toFixed(2)}] → ${imp.fix}`);
|
||||
}
|
||||
}
|
||||
lines.push('');
|
||||
}
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
|
|
|||
72
scripts/kb-eval/score-skill.mjs
Normal file
72
scripts/kb-eval/score-skill.mjs
Normal file
|
|
@ -0,0 +1,72 @@
|
|||
#!/usr/bin/env node
|
||||
// score-skill.mjs — Skill-quality score CLI (Spor D, step 2).
|
||||
//
|
||||
// Runs the eval pipeline (buildReport: deterministic K2/K3/K5/K6/refCount/N1-N5
|
||||
// + cross-skill K10 + merged operator-gated judge cache), scores every skill via
|
||||
// the PURE lib (scoreSkill/scoreReport), and reports a 0-100 quality score with
|
||||
// a sorted improvement list. The corpus invariant is: every skill stays >=90 %.
|
||||
//
|
||||
// Usage:
|
||||
// node scripts/kb-eval/score-skill.mjs # human summary
|
||||
// node scripts/kb-eval/score-skill.mjs --json # machine output (scoreReport)
|
||||
// node scripts/kb-eval/score-skill.mjs --gate 90 # non-zero exit if any skill < 90
|
||||
// node scripts/kb-eval/score-skill.mjs --skill <name> # scope output to one skill
|
||||
// # (still scores the full corpus
|
||||
// # so K10 sibling-overlap is correct)
|
||||
//
|
||||
// K10 requires every sibling description, so even single-skill (incremental)
|
||||
// scoring evaluates the whole corpus and then filters the reported set — this is
|
||||
// the building block for the lifecycle-gate in apply-skill-op.mjs.
|
||||
//
|
||||
// Zero dependencies. Pure scoring lives in lib/skill-score.mjs.
|
||||
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { buildReport } from './eval.mjs';
|
||||
import { scoreReport, formatScoreReport, TARGET } from './lib/skill-score.mjs';
|
||||
|
||||
function argValue(args, flag) {
|
||||
const i = args.indexOf(flag);
|
||||
return i !== -1 && i + 1 < args.length ? args[i + 1] : null;
|
||||
}
|
||||
|
||||
function main() {
|
||||
const args = process.argv.slice(2);
|
||||
const jsonOut = args.includes('--json');
|
||||
const gateRaw = argValue(args, '--gate');
|
||||
const gate = gateRaw !== null ? Number(gateRaw) : null;
|
||||
const skillFilter = argValue(args, '--skill');
|
||||
|
||||
const target = Number.isFinite(gate) ? gate : TARGET;
|
||||
const full = scoreReport(buildReport(), { target });
|
||||
|
||||
// Scope the reported/gated set to one skill if requested (scoring already ran
|
||||
// over the whole corpus, so K10 is computed correctly).
|
||||
let scored = full.scored;
|
||||
if (skillFilter) {
|
||||
scored = scored.filter((s) => s.name === skillFilter);
|
||||
if (scored.length === 0) {
|
||||
process.stderr.write(`score-skill: ukjent skill '${skillFilter}'. Tilgjengelig: ${full.scored.map((s) => s.name).join(', ')}\n`);
|
||||
process.exitCode = 2;
|
||||
return;
|
||||
}
|
||||
}
|
||||
const below = scored.filter((s) => !s.meetsTarget);
|
||||
const result = { target, scored, below };
|
||||
|
||||
if (jsonOut) {
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
} else {
|
||||
process.stdout.write(formatScoreReport(result) + '\n');
|
||||
if (gate !== null) {
|
||||
process.stdout.write(below.length === 0
|
||||
? `✓ Gate ≥${target}: alle ${scored.length} skills består.\n`
|
||||
: `✗ Gate ≥${target}: ${below.length} skill(s) under mål — ${below.map((s) => s.name).join(', ')}.\n`);
|
||||
}
|
||||
}
|
||||
|
||||
if (gate !== null && below.length > 0) process.exitCode = 1;
|
||||
}
|
||||
|
||||
if (process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1]) {
|
||||
main();
|
||||
}
|
||||
113
tests/kb-eval/test-score-skill.test.mjs
Normal file
113
tests/kb-eval/test-score-skill.test.mjs
Normal file
|
|
@ -0,0 +1,113 @@
|
|||
// tests/kb-eval/test-score-skill.test.mjs
|
||||
// Tests for the score-skill CLI building blocks (Spor D, step 2):
|
||||
// - eval.mjs buildReport() assembles the full corpus report (deterministic +
|
||||
// K10 sibling-overlap + merged judge cache).
|
||||
// - lib/skill-score.mjs formatScoreReport() renders the sorted improvement
|
||||
// report (pure).
|
||||
// - the score-skill.mjs CLI: --json emits the scoreReport, --gate sets a
|
||||
// non-zero exit when any skill is below target, --skill scopes the output.
|
||||
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { execFileSync } from 'node:child_process';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { dirname, join } from 'node:path';
|
||||
import { buildReport } from '../../scripts/kb-eval/eval.mjs';
|
||||
import { scoreReport, formatScoreReport } from '../../scripts/kb-eval/lib/skill-score.mjs';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const CLI = join(__dirname, '..', '..', 'scripts', 'kb-eval', 'score-skill.mjs');
|
||||
|
||||
test('buildReport — assembles the full corpus with N1-N5 on every skill', () => {
|
||||
const report = buildReport();
|
||||
assert.ok(Array.isArray(report.skills) && report.skills.length === 5, `got ${report.skills?.length} skills`);
|
||||
for (const s of report.skills) {
|
||||
const d = s.deterministic;
|
||||
assert.ok(d.N1_nameValidity && d.N4_refToc, `${s.name} missing N1/N4`);
|
||||
assert.ok(d.K10_siblingScopeOverlap, `${s.name} missing K10 (sibling overlap not attached)`);
|
||||
}
|
||||
});
|
||||
|
||||
test('scoreReport — over the real corpus returns a score per skill and a below-target set', () => {
|
||||
const result = scoreReport(buildReport());
|
||||
assert.equal(result.target, 90);
|
||||
assert.equal(result.scored.length, 5);
|
||||
for (const s of result.scored) {
|
||||
assert.ok(Number.isFinite(s.score), `${s.name} score not finite`);
|
||||
assert.equal(typeof s.meetsTarget, 'boolean');
|
||||
}
|
||||
// below = subset that does not meet the target
|
||||
assert.ok(result.below.every((s) => !s.meetsTarget));
|
||||
});
|
||||
|
||||
test('scoreReport — a synthetic below-target skill lands in below[]', () => {
|
||||
const synthetic = {
|
||||
skills: [
|
||||
// one perfect-ish, one with a failing K10 floor (capped below 90)
|
||||
{
|
||||
name: 'good', deterministic: {
|
||||
K2_descriptionFormat: { pass: true }, K3_bodyLength: { bodyLines: 100, pass: true },
|
||||
K5_progressiveDisclosure: { namedRatio: 0.5, pass: true }, K6_routingTable: { pass: true },
|
||||
refCountConsistency: { consistent: true }, K10_siblingScopeOverlap: { maxCombined: 1, pass: true, threshold: 7 },
|
||||
N1_nameValidity: { pass: true }, N2_descriptionLength: { pass: true },
|
||||
N3_refNestingDepth: { pass: true }, N4_refToc: { ratio: 1, pass: true, withToc: 1, largeFiles: 1 },
|
||||
N5_forwardSlashPaths: { pass: true },
|
||||
}, judge: null,
|
||||
},
|
||||
{
|
||||
name: 'bad', deterministic: {
|
||||
K2_descriptionFormat: { pass: true }, K3_bodyLength: { bodyLines: 100, pass: true },
|
||||
K5_progressiveDisclosure: { namedRatio: 0.5, pass: true }, K6_routingTable: { pass: true },
|
||||
refCountConsistency: { consistent: true }, K10_siblingScopeOverlap: { maxCombined: 9, pass: false, threshold: 7, worstSibling: 'good' },
|
||||
N1_nameValidity: { pass: true }, N2_descriptionLength: { pass: true },
|
||||
N3_refNestingDepth: { pass: true }, N4_refToc: { ratio: 1, pass: true, withToc: 1, largeFiles: 1 },
|
||||
N5_forwardSlashPaths: { pass: true },
|
||||
}, judge: null,
|
||||
},
|
||||
],
|
||||
};
|
||||
const result = scoreReport(synthetic);
|
||||
assert.equal(result.below.length, 1);
|
||||
assert.equal(result.below[0].name, 'bad');
|
||||
});
|
||||
|
||||
test('formatScoreReport — renders skill names, the target, and improvement fixes', () => {
|
||||
const result = scoreReport(buildReport());
|
||||
const text = formatScoreReport(result);
|
||||
assert.match(text, /ms-ai-advisor/);
|
||||
assert.match(text, /90/); // the target appears
|
||||
// at least one improvement fix line is rendered for a below-target skill
|
||||
if (result.below.length > 0) {
|
||||
assert.ok(/↳|→/.test(text), 'expected improvement bullets in the report');
|
||||
}
|
||||
});
|
||||
|
||||
test('CLI --json — emits a parseable scoreReport with five scored skills (exit 0)', () => {
|
||||
const out = execFileSync('node', [CLI, '--json'], { encoding: 'utf8' });
|
||||
const parsed = JSON.parse(out);
|
||||
assert.equal(parsed.scored.length, 5);
|
||||
assert.equal(parsed.target, 90);
|
||||
});
|
||||
|
||||
test('CLI --skill — scopes the output to a single skill', () => {
|
||||
const out = execFileSync('node', [CLI, '--skill', 'ms-ai-advisor', '--json'], { encoding: 'utf8' });
|
||||
const parsed = JSON.parse(out);
|
||||
assert.equal(parsed.scored.length, 1);
|
||||
assert.equal(parsed.scored[0].name, 'ms-ai-advisor');
|
||||
});
|
||||
|
||||
test('CLI --gate — non-zero exit path: an unreachable target fails the gate (corpus-independent)', () => {
|
||||
let code = 0;
|
||||
try {
|
||||
execFileSync('node', [CLI, '--gate', '101'], { encoding: 'utf8', stdio: 'pipe' });
|
||||
} catch (e) {
|
||||
code = e.status;
|
||||
}
|
||||
assert.equal(code, 1, 'no skill can score 101 -> gate must fail');
|
||||
});
|
||||
|
||||
test('CLI --gate — zero exit path: a target of 0 passes the gate (corpus-independent)', () => {
|
||||
// exits 0; execFileSync throws only on non-zero, so reaching here is the pass
|
||||
execFileSync('node', [CLI, '--gate', '0'], { encoding: 'utf8', stdio: 'pipe' });
|
||||
assert.ok(true);
|
||||
});
|
||||
Loading…
Add table
Add a link
Reference in a new issue