feat(eval): SKAL-1·4b offline gold-scored output eval
Scores committed agent-run fixtures against the golden corpus at (file, rule_key) granularity, building on the deterministic coordinator contract (4a). Offline: committed reviewer payloads, no live agent spawn, no LLM, no network (the LLM-in-the-loop grading is the separate 4c tier). - lib/review/gold-scorer.mjs: scoreFindings (precision/recall/f1 at (file,rule_key) granularity, line+severity ignored) + scoreVerdict; pure, with documented vacuous-set conventions. - tests/fixtures/bakeoff-rich/runs/run-perfect.json: committed run that reproduces all 5 seeded gold findings through runContract. - tests/lib/gold-eval.test.mjs: the scoring RUN (precision/recall/f1 = 1.0, verdict == expected_verdict BLOCK, nothing suppressed/skipped). - lib/util/test-census.mjs: third census category (goldEval) — a scoring run is neither behavior coverage nor a doc-pin; honest-count invariant now 3-way. - docs/eval-corpus/README.md: 4b moved from Future hardening to implemented. Suite 809 -> 822 (820/0/2). gold-scorer covers TP+FP+FN+degenerate paths. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BJQYC5vpkJWxndS55vQQZ6
This commit is contained in:
parent
da418e653d
commit
440594f1b2
7 changed files with 255 additions and 19 deletions
|
|
@ -15,20 +15,21 @@ import { censusTests } from '../../lib/util/test-census.mjs';
|
|||
const HERE = dirname(fileURLToPath(import.meta.url));
|
||||
const TESTS_ROOT = join(HERE, '..');
|
||||
|
||||
test('suite census splits behavior tests from doc-consistency pins (S19)', (t) => {
|
||||
test('suite census splits behavior / doc-pins / gold-eval (S19 + SKAL-1·4b)', (t) => {
|
||||
const c = censusTests(TESTS_ROOT);
|
||||
// Honest-count invariant: the two buckets must account for every top-level
|
||||
// test() declaration — no silent drift between behavior and pin counts.
|
||||
assert.equal(c.behavior + c.docPins, c.total,
|
||||
// Honest-count invariant: the three buckets must account for every top-level
|
||||
// test() declaration — no silent drift between behavior, pin, and eval counts.
|
||||
assert.equal(c.behavior + c.docPins + c.goldEval, c.total,
|
||||
'census buckets must sum to the total declaration count');
|
||||
assert.ok(c.docPins > 0, 'doc-consistency pin bucket must be non-empty (regex/glob sanity)');
|
||||
assert.ok(c.behavior > 0, 'behavior bucket must be non-empty (regex/glob sanity)');
|
||||
assert.ok(c.goldEval > 0, 'gold-eval scoring-run bucket must be non-empty (SKAL-1·4b present)');
|
||||
// Report the split so the cited count is honest (audit §Top changes #8).
|
||||
// Metric = top-level test() declarations; node:test's runtime total counts
|
||||
// subtests too and is therefore ≥ this number.
|
||||
t.diagnostic(
|
||||
`behavior=${c.behavior} doc-consistency-pins=${c.docPins} ` +
|
||||
`total=${c.total} (top-level test() declarations)`,
|
||||
`gold-eval=${c.goldEval} total=${c.total} (top-level test() declarations)`,
|
||||
);
|
||||
});
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue