llm-ingestion-okf/tests/test_adjudicate.py
Kjell Tore Guttormsen 62b61927a4 fix(adjudicate): a judgement that keeps nothing gets an artifact
Measured on the K3 corpus: 4 of 12 judgements produced no artifact, because
the verdict was "none of these segments should be persisted" and the plan
grammar refuses zero entries. That refusal is correct for the run path -- an
empty plan replayed would silently persist nothing for a document that was
dropped -- so the grammar is untouched and the recording tool is taught to
record a rejection instead.

Refusing to materialize and refusing to record are different acts. The
rejection artifact is deliberately NOT replayable: parse_segmentation_plan
still refuses it, and the suite asserts that rather than assuming it. The dwell
time rides at the top level because there is no entry to carry it, and a
ratified rejection with no time on it is as unfalsifiable as a ratified
acceptance with none.

Only the empty LIST takes the branch. A missing entries key, or one that is not
a list, stays the grammar's to refuse: "the adjudicator kept nothing" and
"this file is not a plan" must not collapse.

K4a re-run after the change: propose, adjudicate, run the path twice into two
bundles under SEGMENTED_OKF_V0_2, diff -r exit 0 with no output. 1034 -> 1041
tests.
2026-09-02 16:15:28 +02:00

345 lines
12 KiB
Python

"""The adjudication command: it records a judgement, and never invents one.
A proposal a human has not looked at must never be replayable as an
adjudication, because replay is exactly what the run path does with a plan --
deterministically and forever. So this command writes a SIBLING record and
leaves the proposal untouched: the two files together say who judged what,
when, and how long it took, and either can be re-read against the other.
Three properties are pinned here rather than described:
- **The model leg is OFF by default.** Pre-annotation has been measured
LOWERING a good annotator's accuracy, from 98.1 % to 95.8 %, so a leg that
cannot be switched off is a leg whose value can never be measured. With it
off, no process is spawned at all -- asserted by breaking `subprocess.run`.
- **The CLI is named, and the other one is excluded BY NAME.** The model leg
shells out to the `claude` CLI. `gemini` is not merely unmentioned; its
absence from the module is a test, because "we did not use it" and "nothing
stops us using it" look identical in a review.
- **Dwell time travels with the verdict** (PM decision B2). A ratified flag
with no per-item time is unfalsifiable, and it is the same number that makes
adjudication throughput measurable at all.
It lives outside `src/`, so it never enters a wheel and no consumer's install
surface changes because it exists.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
from typing import Any
import pytest
from llm_ingestion_okf.errors import SegmentationError
from llm_ingestion_okf.segmentation import parse_segmentation_plan
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "tools"))
import okf_adjudicate # noqa: E402
import okf_propose_segments # noqa: E402
DOCUMENT = """# N500 Vegbygging
Innledende tekst om vegbygging og dens omfang.
## 3.1 Brannkonsept
Krav til seksjonering av bygget.
## 3.2 Roemning
To uavhengige roemningsveier.
"""
ADJUDICATOR = "ktg"
AT = "2026-09-02T10:00:00Z"
def proposal(tmp_path: Path) -> Path:
source = tmp_path / "n500.md"
source.write_text(DOCUMENT, encoding="utf-8", newline="")
out = tmp_path / "plan.json"
assert okf_propose_segments.main([str(source), "--out", str(out), "--proposed-at", AT]) == 0
return out
def adjudicate(tmp_path: Path, *extra: str) -> tuple[int, Path]:
verdict = tmp_path / "adjudicated.json"
code = okf_adjudicate.main(
[
"--plan",
str(proposal(tmp_path)),
"--out",
str(verdict),
"--adjudicator",
ADJUDICATOR,
"--adjudicated-at",
AT,
*extra,
]
)
return code, verdict
def payload(path: Path) -> dict[str, Any]:
return json.loads(path.read_text(encoding="utf-8"))
def test_the_proposal_survives_untouched(tmp_path: Path) -> None:
plan_path = proposal(tmp_path)
before = plan_path.read_bytes()
okf_adjudicate.main(
[
"--plan",
str(plan_path),
"--out",
str(tmp_path / "adjudicated.json"),
"--adjudicator",
ADJUDICATOR,
"--adjudicated-at",
AT,
]
)
assert plan_path.read_bytes() == before
def test_the_verdict_records_adjudicator_timestamp_and_dwell(tmp_path: Path) -> None:
code, verdict = adjudicate(tmp_path)
assert code == 0
written = payload(verdict)
assert written["adjudicated"] is True
for entry in written["entries"]:
record = entry["adjudication"]
assert record["adjudicated_by"] == ADJUDICATOR
assert record["adjudicated_at"] == AT
assert isinstance(record["adjudication_dwell_s"], int)
assert not isinstance(record["adjudication_dwell_s"], bool)
def test_the_verdict_parses_as_a_segmentation_plan(tmp_path: Path) -> None:
_, verdict = adjudicate(tmp_path)
parsed = parse_segmentation_plan(payload(verdict))
assert parsed.adjudicated is True
assert all(entry.adjudication is not None for entry in parsed.entries)
def test_replaying_the_same_verdict_produces_identical_bytes(tmp_path: Path) -> None:
"""K4a's mechanism: an adjudication is data, so a re-run is a copy."""
_, first = adjudicate(tmp_path)
kept = first.read_bytes()
second = tmp_path / "again.json"
okf_adjudicate.main(
[
"--plan",
str(tmp_path / "plan.json"),
"--out",
str(second),
"--adjudicator",
ADJUDICATOR,
"--adjudicated-at",
AT,
]
)
assert second.read_bytes() == kept
def test_with_the_model_leg_off_no_process_is_spawned(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""Asserted by BREAKING the spawn, not by reading the code.
A test that merely inspects the default would pass just as happily if the
default were ignored.
"""
def refuse(*args: object, **kwargs: object) -> None:
raise AssertionError("the model leg spawned a process while switched off")
monkeypatch.setattr(subprocess, "run", refuse)
code, _ = adjudicate(tmp_path)
assert code == 0
def test_the_model_leg_is_off_unless_asked_for(tmp_path: Path) -> None:
assert (
okf_adjudicate.parse_args(
["--plan", "p", "--out", "o", "--adjudicator", "a", "--adjudicated-at", AT]
).model
is None
)
def test_the_resolved_argv_starts_with_the_claude_binary() -> None:
argv = okf_adjudicate.model_argv("claude-opus-5", "spoersmaal")
assert argv[0] == okf_adjudicate.CLAUDE_CLI
assert Path(argv[0]).name == "claude"
assert "--model" in argv
assert argv[argv.index("--model") + 1] == "claude-opus-5"
def test_the_other_cli_is_excluded_by_name_not_merely_unused() -> None:
""" "We did not use it" and "nothing stops us using it" look identical in a
review. This is the difference, as a measurement."""
module = Path(okf_adjudicate.__file__).read_text(encoding="utf-8")
assert "gemini" not in module.lower()
def test_the_gemini_check_can_actually_fire() -> None:
"""The negative control for the check above: prove it can find the word."""
assert "gemini" in "a line naming gemini".lower()
def test_a_missing_plan_exits_two_and_says_so(
tmp_path: Path, capsys: pytest.CaptureFixture[str]
) -> None:
code = okf_adjudicate.main(
[
"--plan",
str(tmp_path / "nothing.json"),
"--out",
str(tmp_path / "out.json"),
"--adjudicator",
ADJUDICATOR,
"--adjudicated-at",
AT,
]
)
assert code == 2
assert "nothing.json" in capsys.readouterr().err
# --- the empty plan: a judgement with nothing to keep -----------------------
#
# Measured on the K3 corpus: 4 of 12 judgements produced no artifact at all,
# because the adjudicator's verdict was "none of these segments should be
# persisted" and the parser refuses a plan with zero entries. That refusal is
# CORRECT for the run path -- an empty plan would silently persist nothing for a
# document that was dropped -- so the grammar is left alone and the recording
# tool is taught to record a rejection. The two are different acts: refusing to
# materialize is about a bundle, recording a judgement is about a person.
def empty_proposal(tmp_path: Path) -> Path:
"""A real proposal with its entries removed -- the plan-level fields stay
exactly as the proposer wrote them, so this is a rejection and not a stub."""
plan_path = proposal(tmp_path)
written = payload(plan_path)
written["entries"] = []
rejected = tmp_path / "rejected.json"
rejected.write_text(json.dumps(written, indent=2) + "\n", encoding="utf-8", newline="")
return rejected
def adjudicate_empty(
tmp_path: Path, plan_path: Path, out_name: str = "verdict.json"
) -> tuple[int, Path]:
verdict = tmp_path / out_name
code = okf_adjudicate.main(
[
"--plan",
str(plan_path),
"--out",
str(verdict),
"--adjudicator",
ADJUDICATOR,
"--adjudicated-at",
AT,
]
)
return code, verdict
def test_a_judgement_over_an_empty_plan_gets_an_artifact(tmp_path: Path) -> None:
"""The defect this closes: the judgement happened and left no trace."""
code, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
assert code == 0
assert verdict.is_file()
written = payload(verdict)
assert written["entries"] == []
assert written["adjudicated"] is True
assert written["adjudicated_by"] == ADJUDICATOR
assert written["adjudicated_at"] == AT
def test_the_empty_verdict_carries_the_dwell_time_at_the_top(tmp_path: Path) -> None:
"""There is no entry to hang it on, and a ratified rejection with no time on
it is as unfalsifiable as a ratified acceptance with none."""
_, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
written = payload(verdict)
assert isinstance(written["adjudication_dwell_s"], int)
assert not isinstance(written["adjudication_dwell_s"], bool)
assert written["adjudication_dwell_s"] > 0
def test_the_empty_verdict_is_not_replayable_by_the_run_path(tmp_path: Path) -> None:
"""The grammar is UNCHANGED. Recording a rejection and materializing from it
are different acts, and only the first one is now possible."""
_, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
with pytest.raises(SegmentationError) as excinfo:
parse_segmentation_plan(payload(verdict))
assert excinfo.value.code == "segmentation_plan_invalid"
def test_the_rejected_proposal_survives_untouched(tmp_path: Path) -> None:
plan_path = empty_proposal(tmp_path)
before = plan_path.read_bytes()
adjudicate_empty(tmp_path, plan_path)
assert plan_path.read_bytes() == before
def test_replaying_an_empty_verdict_produces_identical_bytes(tmp_path: Path) -> None:
"""K4a over the arm that had no artifact to compare before."""
plan_path = empty_proposal(tmp_path)
_, first = adjudicate_empty(tmp_path, plan_path, "first.json")
kept = first.read_bytes()
_, second = adjudicate_empty(tmp_path, plan_path, "second.json")
assert second.read_bytes() == kept
def test_an_empty_plan_missing_a_required_field_is_still_refused(
tmp_path: Path, capsys: pytest.CaptureFixture[str]
) -> None:
"""The empty branch is not a hole in the validation: a plan is still a plan,
and only its entry list is allowed to be empty."""
plan_path = empty_proposal(tmp_path)
written = payload(plan_path)
del written["source_sha256"]
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
code, verdict = adjudicate_empty(tmp_path, plan_path)
assert code == 2
assert "source_sha256" in capsys.readouterr().err
assert not verdict.exists()
def test_an_entries_value_that_is_not_a_list_is_still_refused(tmp_path: Path) -> None:
"""Empty is a judgement; the wrong TYPE is a malformed plan, and the two
must not collapse. The malformed one still meets the unchanged grammar.
It reaches the caller as a raised SegmentationError rather than as exit 2,
which is pre-existing behaviour for every malformed plan and is left alone
here rather than repaired inside a change about empty ones. Recorded as a
finding, not fixed."""
plan_path = empty_proposal(tmp_path)
written = payload(plan_path)
written["entries"] = "none"
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
with pytest.raises(SegmentationError) as excinfo:
adjudicate_empty(tmp_path, plan_path)
assert excinfo.value.code == "segmentation_plan_invalid"
assert not (tmp_path / "verdict.json").exists()