Measured on the K3 corpus: 4 of 12 judgements produced no artifact, because the verdict was "none of these segments should be persisted" and the plan grammar refuses zero entries. That refusal is correct for the run path -- an empty plan replayed would silently persist nothing for a document that was dropped -- so the grammar is untouched and the recording tool is taught to record a rejection instead. Refusing to materialize and refusing to record are different acts. The rejection artifact is deliberately NOT replayable: parse_segmentation_plan still refuses it, and the suite asserts that rather than assuming it. The dwell time rides at the top level because there is no entry to carry it, and a ratified rejection with no time on it is as unfalsifiable as a ratified acceptance with none. Only the empty LIST takes the branch. A missing entries key, or one that is not a list, stays the grammar's to refuse: "the adjudicator kept nothing" and "this file is not a plan" must not collapse. K4a re-run after the change: propose, adjudicate, run the path twice into two bundles under SEGMENTED_OKF_V0_2, diff -r exit 0 with no output. 1034 -> 1041 tests.
345 lines
12 KiB
Python
345 lines
12 KiB
Python
"""The adjudication command: it records a judgement, and never invents one.
|
|
|
|
A proposal a human has not looked at must never be replayable as an
|
|
adjudication, because replay is exactly what the run path does with a plan --
|
|
deterministically and forever. So this command writes a SIBLING record and
|
|
leaves the proposal untouched: the two files together say who judged what,
|
|
when, and how long it took, and either can be re-read against the other.
|
|
|
|
Three properties are pinned here rather than described:
|
|
|
|
- **The model leg is OFF by default.** Pre-annotation has been measured
|
|
LOWERING a good annotator's accuracy, from 98.1 % to 95.8 %, so a leg that
|
|
cannot be switched off is a leg whose value can never be measured. With it
|
|
off, no process is spawned at all -- asserted by breaking `subprocess.run`.
|
|
- **The CLI is named, and the other one is excluded BY NAME.** The model leg
|
|
shells out to the `claude` CLI. `gemini` is not merely unmentioned; its
|
|
absence from the module is a test, because "we did not use it" and "nothing
|
|
stops us using it" look identical in a review.
|
|
- **Dwell time travels with the verdict** (PM decision B2). A ratified flag
|
|
with no per-item time is unfalsifiable, and it is the same number that makes
|
|
adjudication throughput measurable at all.
|
|
|
|
It lives outside `src/`, so it never enters a wheel and no consumer's install
|
|
surface changes because it exists.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf.errors import SegmentationError
|
|
from llm_ingestion_okf.segmentation import parse_segmentation_plan
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "tools"))
|
|
|
|
import okf_adjudicate # noqa: E402
|
|
import okf_propose_segments # noqa: E402
|
|
|
|
DOCUMENT = """# N500 Vegbygging
|
|
|
|
Innledende tekst om vegbygging og dens omfang.
|
|
|
|
## 3.1 Brannkonsept
|
|
|
|
Krav til seksjonering av bygget.
|
|
|
|
## 3.2 Roemning
|
|
|
|
To uavhengige roemningsveier.
|
|
"""
|
|
|
|
ADJUDICATOR = "ktg"
|
|
AT = "2026-09-02T10:00:00Z"
|
|
|
|
|
|
def proposal(tmp_path: Path) -> Path:
|
|
source = tmp_path / "n500.md"
|
|
source.write_text(DOCUMENT, encoding="utf-8", newline="")
|
|
out = tmp_path / "plan.json"
|
|
assert okf_propose_segments.main([str(source), "--out", str(out), "--proposed-at", AT]) == 0
|
|
return out
|
|
|
|
|
|
def adjudicate(tmp_path: Path, *extra: str) -> tuple[int, Path]:
|
|
verdict = tmp_path / "adjudicated.json"
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(proposal(tmp_path)),
|
|
"--out",
|
|
str(verdict),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
*extra,
|
|
]
|
|
)
|
|
return code, verdict
|
|
|
|
|
|
def payload(path: Path) -> dict[str, Any]:
|
|
return json.loads(path.read_text(encoding="utf-8"))
|
|
|
|
|
|
def test_the_proposal_survives_untouched(tmp_path: Path) -> None:
|
|
plan_path = proposal(tmp_path)
|
|
before = plan_path.read_bytes()
|
|
okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(plan_path),
|
|
"--out",
|
|
str(tmp_path / "adjudicated.json"),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
assert plan_path.read_bytes() == before
|
|
|
|
|
|
def test_the_verdict_records_adjudicator_timestamp_and_dwell(tmp_path: Path) -> None:
|
|
code, verdict = adjudicate(tmp_path)
|
|
assert code == 0
|
|
written = payload(verdict)
|
|
assert written["adjudicated"] is True
|
|
for entry in written["entries"]:
|
|
record = entry["adjudication"]
|
|
assert record["adjudicated_by"] == ADJUDICATOR
|
|
assert record["adjudicated_at"] == AT
|
|
assert isinstance(record["adjudication_dwell_s"], int)
|
|
assert not isinstance(record["adjudication_dwell_s"], bool)
|
|
|
|
|
|
def test_the_verdict_parses_as_a_segmentation_plan(tmp_path: Path) -> None:
|
|
_, verdict = adjudicate(tmp_path)
|
|
parsed = parse_segmentation_plan(payload(verdict))
|
|
assert parsed.adjudicated is True
|
|
assert all(entry.adjudication is not None for entry in parsed.entries)
|
|
|
|
|
|
def test_replaying_the_same_verdict_produces_identical_bytes(tmp_path: Path) -> None:
|
|
"""K4a's mechanism: an adjudication is data, so a re-run is a copy."""
|
|
_, first = adjudicate(tmp_path)
|
|
kept = first.read_bytes()
|
|
second = tmp_path / "again.json"
|
|
okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(tmp_path / "plan.json"),
|
|
"--out",
|
|
str(second),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
assert second.read_bytes() == kept
|
|
|
|
|
|
def test_with_the_model_leg_off_no_process_is_spawned(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
"""Asserted by BREAKING the spawn, not by reading the code.
|
|
|
|
A test that merely inspects the default would pass just as happily if the
|
|
default were ignored.
|
|
"""
|
|
|
|
def refuse(*args: object, **kwargs: object) -> None:
|
|
raise AssertionError("the model leg spawned a process while switched off")
|
|
|
|
monkeypatch.setattr(subprocess, "run", refuse)
|
|
code, _ = adjudicate(tmp_path)
|
|
assert code == 0
|
|
|
|
|
|
def test_the_model_leg_is_off_unless_asked_for(tmp_path: Path) -> None:
|
|
assert (
|
|
okf_adjudicate.parse_args(
|
|
["--plan", "p", "--out", "o", "--adjudicator", "a", "--adjudicated-at", AT]
|
|
).model
|
|
is None
|
|
)
|
|
|
|
|
|
def test_the_resolved_argv_starts_with_the_claude_binary() -> None:
|
|
argv = okf_adjudicate.model_argv("claude-opus-5", "spoersmaal")
|
|
assert argv[0] == okf_adjudicate.CLAUDE_CLI
|
|
assert Path(argv[0]).name == "claude"
|
|
assert "--model" in argv
|
|
assert argv[argv.index("--model") + 1] == "claude-opus-5"
|
|
|
|
|
|
def test_the_other_cli_is_excluded_by_name_not_merely_unused() -> None:
|
|
""" "We did not use it" and "nothing stops us using it" look identical in a
|
|
review. This is the difference, as a measurement."""
|
|
module = Path(okf_adjudicate.__file__).read_text(encoding="utf-8")
|
|
assert "gemini" not in module.lower()
|
|
|
|
|
|
def test_the_gemini_check_can_actually_fire() -> None:
|
|
"""The negative control for the check above: prove it can find the word."""
|
|
assert "gemini" in "a line naming gemini".lower()
|
|
|
|
|
|
def test_a_missing_plan_exits_two_and_says_so(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(tmp_path / "nothing.json"),
|
|
"--out",
|
|
str(tmp_path / "out.json"),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
assert code == 2
|
|
assert "nothing.json" in capsys.readouterr().err
|
|
|
|
|
|
# --- the empty plan: a judgement with nothing to keep -----------------------
|
|
#
|
|
# Measured on the K3 corpus: 4 of 12 judgements produced no artifact at all,
|
|
# because the adjudicator's verdict was "none of these segments should be
|
|
# persisted" and the parser refuses a plan with zero entries. That refusal is
|
|
# CORRECT for the run path -- an empty plan would silently persist nothing for a
|
|
# document that was dropped -- so the grammar is left alone and the recording
|
|
# tool is taught to record a rejection. The two are different acts: refusing to
|
|
# materialize is about a bundle, recording a judgement is about a person.
|
|
|
|
|
|
def empty_proposal(tmp_path: Path) -> Path:
|
|
"""A real proposal with its entries removed -- the plan-level fields stay
|
|
exactly as the proposer wrote them, so this is a rejection and not a stub."""
|
|
plan_path = proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
written["entries"] = []
|
|
rejected = tmp_path / "rejected.json"
|
|
rejected.write_text(json.dumps(written, indent=2) + "\n", encoding="utf-8", newline="")
|
|
return rejected
|
|
|
|
|
|
def adjudicate_empty(
|
|
tmp_path: Path, plan_path: Path, out_name: str = "verdict.json"
|
|
) -> tuple[int, Path]:
|
|
verdict = tmp_path / out_name
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(plan_path),
|
|
"--out",
|
|
str(verdict),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
return code, verdict
|
|
|
|
|
|
def test_a_judgement_over_an_empty_plan_gets_an_artifact(tmp_path: Path) -> None:
|
|
"""The defect this closes: the judgement happened and left no trace."""
|
|
code, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
|
|
|
|
assert code == 0
|
|
assert verdict.is_file()
|
|
written = payload(verdict)
|
|
assert written["entries"] == []
|
|
assert written["adjudicated"] is True
|
|
assert written["adjudicated_by"] == ADJUDICATOR
|
|
assert written["adjudicated_at"] == AT
|
|
|
|
|
|
def test_the_empty_verdict_carries_the_dwell_time_at_the_top(tmp_path: Path) -> None:
|
|
"""There is no entry to hang it on, and a ratified rejection with no time on
|
|
it is as unfalsifiable as a ratified acceptance with none."""
|
|
_, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
|
|
written = payload(verdict)
|
|
|
|
assert isinstance(written["adjudication_dwell_s"], int)
|
|
assert not isinstance(written["adjudication_dwell_s"], bool)
|
|
assert written["adjudication_dwell_s"] > 0
|
|
|
|
|
|
def test_the_empty_verdict_is_not_replayable_by_the_run_path(tmp_path: Path) -> None:
|
|
"""The grammar is UNCHANGED. Recording a rejection and materializing from it
|
|
are different acts, and only the first one is now possible."""
|
|
_, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
|
|
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
parse_segmentation_plan(payload(verdict))
|
|
|
|
assert excinfo.value.code == "segmentation_plan_invalid"
|
|
|
|
|
|
def test_the_rejected_proposal_survives_untouched(tmp_path: Path) -> None:
|
|
plan_path = empty_proposal(tmp_path)
|
|
before = plan_path.read_bytes()
|
|
|
|
adjudicate_empty(tmp_path, plan_path)
|
|
|
|
assert plan_path.read_bytes() == before
|
|
|
|
|
|
def test_replaying_an_empty_verdict_produces_identical_bytes(tmp_path: Path) -> None:
|
|
"""K4a over the arm that had no artifact to compare before."""
|
|
plan_path = empty_proposal(tmp_path)
|
|
_, first = adjudicate_empty(tmp_path, plan_path, "first.json")
|
|
kept = first.read_bytes()
|
|
|
|
_, second = adjudicate_empty(tmp_path, plan_path, "second.json")
|
|
|
|
assert second.read_bytes() == kept
|
|
|
|
|
|
def test_an_empty_plan_missing_a_required_field_is_still_refused(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
"""The empty branch is not a hole in the validation: a plan is still a plan,
|
|
and only its entry list is allowed to be empty."""
|
|
plan_path = empty_proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
del written["source_sha256"]
|
|
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
|
|
|
|
code, verdict = adjudicate_empty(tmp_path, plan_path)
|
|
|
|
assert code == 2
|
|
assert "source_sha256" in capsys.readouterr().err
|
|
assert not verdict.exists()
|
|
|
|
|
|
def test_an_entries_value_that_is_not_a_list_is_still_refused(tmp_path: Path) -> None:
|
|
"""Empty is a judgement; the wrong TYPE is a malformed plan, and the two
|
|
must not collapse. The malformed one still meets the unchanged grammar.
|
|
|
|
It reaches the caller as a raised SegmentationError rather than as exit 2,
|
|
which is pre-existing behaviour for every malformed plan and is left alone
|
|
here rather than repaired inside a change about empty ones. Recorded as a
|
|
finding, not fixed."""
|
|
plan_path = empty_proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
written["entries"] = "none"
|
|
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
|
|
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
adjudicate_empty(tmp_path, plan_path)
|
|
|
|
assert excinfo.value.code == "segmentation_plan_invalid"
|
|
assert not (tmp_path / "verdict.json").exists()
|