run() parses the plan before building anything and parses its own
verdict after. Only the first failure is the operator's file.
The second is reachable: an empty --adjudicator produces a verdict the
grammar refuses ("adjudication field 'adjudicated_by' must be a
non-empty string"). The plan parsed fine; the fault is in what this
command stamped onto it.
Catching SegmentationError at the top of main() -- the previous commit
-- caught both raise sites and printed "malformed plan" for each. On
this path that is a clean, confident, WRONG diagnosis: it sends the
operator to fix the one artifact that was fine. Worse than the traceback
it replaced, because a traceback at least does not claim to know.
The verdict parse now raises AdjudicationError, which is what "this
command failed" already means in this file and already returns 2. Exit
code unchanged either way, nothing written either way; only the message
changes.
Suite 1073 -> 1074 passed (pytest exit 0, measured without a pipe);
ruff and mypy --strict clean.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
415 lines
14 KiB
Python
415 lines
14 KiB
Python
"""The adjudication command: it records a judgement, and never invents one.
|
|
|
|
A proposal a human has not looked at must never be replayable as an
|
|
adjudication, because replay is exactly what the run path does with a plan --
|
|
deterministically and forever. So this command writes a SIBLING record and
|
|
leaves the proposal untouched: the two files together say who judged what,
|
|
when, and how long it took, and either can be re-read against the other.
|
|
|
|
Three properties are pinned here rather than described:
|
|
|
|
- **The model leg is OFF by default.** Pre-annotation has been measured
|
|
LOWERING a good annotator's accuracy, from 98.1 % to 95.8 %, so a leg that
|
|
cannot be switched off is a leg whose value can never be measured. With it
|
|
off, no process is spawned at all -- asserted by breaking `subprocess.run`.
|
|
- **The CLI is named, and the other one is excluded BY NAME.** The model leg
|
|
shells out to the `claude` CLI. `gemini` is not merely unmentioned; its
|
|
absence from the module is a test, because "we did not use it" and "nothing
|
|
stops us using it" look identical in a review.
|
|
- **Dwell time travels with the verdict** (PM decision B2). A ratified flag
|
|
with no per-item time is unfalsifiable, and it is the same number that makes
|
|
adjudication throughput measurable at all.
|
|
|
|
It lives outside `src/`, so it never enters a wheel and no consumer's install
|
|
surface changes because it exists.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf.errors import SegmentationError
|
|
from llm_ingestion_okf.segmentation import parse_segmentation_plan
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "tools"))
|
|
|
|
import okf_adjudicate # noqa: E402
|
|
import okf_propose_segments # noqa: E402
|
|
|
|
DOCUMENT = """# N500 Vegbygging
|
|
|
|
Innledende tekst om vegbygging og dens omfang.
|
|
|
|
## 3.1 Brannkonsept
|
|
|
|
Krav til seksjonering av bygget.
|
|
|
|
## 3.2 Roemning
|
|
|
|
To uavhengige roemningsveier.
|
|
"""
|
|
|
|
ADJUDICATOR = "ktg"
|
|
AT = "2026-09-02T10:00:00Z"
|
|
|
|
|
|
def proposal(tmp_path: Path) -> Path:
|
|
source = tmp_path / "n500.md"
|
|
source.write_text(DOCUMENT, encoding="utf-8", newline="")
|
|
out = tmp_path / "plan.json"
|
|
assert okf_propose_segments.main([str(source), "--out", str(out), "--proposed-at", AT]) == 0
|
|
return out
|
|
|
|
|
|
def adjudicate(tmp_path: Path, *extra: str) -> tuple[int, Path]:
|
|
verdict = tmp_path / "adjudicated.json"
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(proposal(tmp_path)),
|
|
"--out",
|
|
str(verdict),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
*extra,
|
|
]
|
|
)
|
|
return code, verdict
|
|
|
|
|
|
def payload(path: Path) -> dict[str, Any]:
|
|
return json.loads(path.read_text(encoding="utf-8"))
|
|
|
|
|
|
def test_the_proposal_survives_untouched(tmp_path: Path) -> None:
|
|
plan_path = proposal(tmp_path)
|
|
before = plan_path.read_bytes()
|
|
okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(plan_path),
|
|
"--out",
|
|
str(tmp_path / "adjudicated.json"),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
assert plan_path.read_bytes() == before
|
|
|
|
|
|
def test_the_verdict_records_adjudicator_timestamp_and_dwell(tmp_path: Path) -> None:
|
|
code, verdict = adjudicate(tmp_path)
|
|
assert code == 0
|
|
written = payload(verdict)
|
|
assert written["adjudicated"] is True
|
|
for entry in written["entries"]:
|
|
record = entry["adjudication"]
|
|
assert record["adjudicated_by"] == ADJUDICATOR
|
|
assert record["adjudicated_at"] == AT
|
|
assert isinstance(record["adjudication_dwell_s"], int)
|
|
assert not isinstance(record["adjudication_dwell_s"], bool)
|
|
|
|
|
|
def test_the_verdict_parses_as_a_segmentation_plan(tmp_path: Path) -> None:
|
|
_, verdict = adjudicate(tmp_path)
|
|
parsed = parse_segmentation_plan(payload(verdict))
|
|
assert parsed.adjudicated is True
|
|
assert all(entry.adjudication is not None for entry in parsed.entries)
|
|
|
|
|
|
def test_replaying_the_same_verdict_produces_identical_bytes(tmp_path: Path) -> None:
|
|
"""K4a's mechanism: an adjudication is data, so a re-run is a copy."""
|
|
_, first = adjudicate(tmp_path)
|
|
kept = first.read_bytes()
|
|
second = tmp_path / "again.json"
|
|
okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(tmp_path / "plan.json"),
|
|
"--out",
|
|
str(second),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
assert second.read_bytes() == kept
|
|
|
|
|
|
def test_with_the_model_leg_off_no_process_is_spawned(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
"""Asserted by BREAKING the spawn, not by reading the code.
|
|
|
|
A test that merely inspects the default would pass just as happily if the
|
|
default were ignored.
|
|
"""
|
|
|
|
def refuse(*args: object, **kwargs: object) -> None:
|
|
raise AssertionError("the model leg spawned a process while switched off")
|
|
|
|
monkeypatch.setattr(subprocess, "run", refuse)
|
|
code, _ = adjudicate(tmp_path)
|
|
assert code == 0
|
|
|
|
|
|
def test_the_model_leg_is_off_unless_asked_for(tmp_path: Path) -> None:
|
|
assert (
|
|
okf_adjudicate.parse_args(
|
|
["--plan", "p", "--out", "o", "--adjudicator", "a", "--adjudicated-at", AT]
|
|
).model
|
|
is None
|
|
)
|
|
|
|
|
|
def test_the_resolved_argv_starts_with_the_claude_binary() -> None:
|
|
argv = okf_adjudicate.model_argv("claude-opus-5", "spoersmaal")
|
|
assert argv[0] == okf_adjudicate.CLAUDE_CLI
|
|
assert Path(argv[0]).name == "claude"
|
|
assert "--model" in argv
|
|
assert argv[argv.index("--model") + 1] == "claude-opus-5"
|
|
|
|
|
|
def test_the_other_cli_is_excluded_by_name_not_merely_unused() -> None:
|
|
""" "We did not use it" and "nothing stops us using it" look identical in a
|
|
review. This is the difference, as a measurement."""
|
|
module = Path(okf_adjudicate.__file__).read_text(encoding="utf-8")
|
|
assert "gemini" not in module.lower()
|
|
|
|
|
|
def test_the_gemini_check_can_actually_fire() -> None:
|
|
"""The negative control for the check above: prove it can find the word."""
|
|
assert "gemini" in "a line naming gemini".lower()
|
|
|
|
|
|
def test_a_missing_plan_exits_two_and_says_so(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(tmp_path / "nothing.json"),
|
|
"--out",
|
|
str(tmp_path / "out.json"),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
assert code == 2
|
|
assert "nothing.json" in capsys.readouterr().err
|
|
|
|
|
|
# --- the empty plan: a judgement with nothing to keep -----------------------
|
|
#
|
|
# Measured on the K3 corpus: 4 of 12 judgements produced no artifact at all,
|
|
# because the adjudicator's verdict was "none of these segments should be
|
|
# persisted" and the parser refuses a plan with zero entries. That refusal is
|
|
# CORRECT for the run path -- an empty plan would silently persist nothing for a
|
|
# document that was dropped -- so the grammar is left alone and the recording
|
|
# tool is taught to record a rejection. The two are different acts: refusing to
|
|
# materialize is about a bundle, recording a judgement is about a person.
|
|
|
|
|
|
def empty_proposal(tmp_path: Path) -> Path:
|
|
"""A real proposal with its entries removed -- the plan-level fields stay
|
|
exactly as the proposer wrote them, so this is a rejection and not a stub."""
|
|
plan_path = proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
written["entries"] = []
|
|
rejected = tmp_path / "rejected.json"
|
|
rejected.write_text(json.dumps(written, indent=2) + "\n", encoding="utf-8", newline="")
|
|
return rejected
|
|
|
|
|
|
def adjudicate_empty(
|
|
tmp_path: Path, plan_path: Path, out_name: str = "verdict.json"
|
|
) -> tuple[int, Path]:
|
|
verdict = tmp_path / out_name
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(plan_path),
|
|
"--out",
|
|
str(verdict),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
return code, verdict
|
|
|
|
|
|
def test_a_judgement_over_an_empty_plan_gets_an_artifact(tmp_path: Path) -> None:
|
|
"""The defect this closes: the judgement happened and left no trace."""
|
|
code, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
|
|
|
|
assert code == 0
|
|
assert verdict.is_file()
|
|
written = payload(verdict)
|
|
assert written["entries"] == []
|
|
assert written["adjudicated"] is True
|
|
assert written["adjudicated_by"] == ADJUDICATOR
|
|
assert written["adjudicated_at"] == AT
|
|
|
|
|
|
def test_the_empty_verdict_carries_the_dwell_time_at_the_top(tmp_path: Path) -> None:
|
|
"""There is no entry to hang it on, and a ratified rejection with no time on
|
|
it is as unfalsifiable as a ratified acceptance with none."""
|
|
_, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
|
|
written = payload(verdict)
|
|
|
|
assert isinstance(written["adjudication_dwell_s"], int)
|
|
assert not isinstance(written["adjudication_dwell_s"], bool)
|
|
assert written["adjudication_dwell_s"] > 0
|
|
|
|
|
|
def test_the_empty_verdict_is_not_replayable_by_the_run_path(tmp_path: Path) -> None:
|
|
"""The grammar is UNCHANGED. Recording a rejection and materializing from it
|
|
are different acts, and only the first one is now possible."""
|
|
_, verdict = adjudicate_empty(tmp_path, empty_proposal(tmp_path))
|
|
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
parse_segmentation_plan(payload(verdict))
|
|
|
|
assert excinfo.value.code == "segmentation_plan_invalid"
|
|
|
|
|
|
def test_the_rejected_proposal_survives_untouched(tmp_path: Path) -> None:
|
|
plan_path = empty_proposal(tmp_path)
|
|
before = plan_path.read_bytes()
|
|
|
|
adjudicate_empty(tmp_path, plan_path)
|
|
|
|
assert plan_path.read_bytes() == before
|
|
|
|
|
|
def test_replaying_an_empty_verdict_produces_identical_bytes(tmp_path: Path) -> None:
|
|
"""K4a over the arm that had no artifact to compare before."""
|
|
plan_path = empty_proposal(tmp_path)
|
|
_, first = adjudicate_empty(tmp_path, plan_path, "first.json")
|
|
kept = first.read_bytes()
|
|
|
|
_, second = adjudicate_empty(tmp_path, plan_path, "second.json")
|
|
|
|
assert second.read_bytes() == kept
|
|
|
|
|
|
def test_an_empty_plan_missing_a_required_field_is_still_refused(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
"""The empty branch is not a hole in the validation: a plan is still a plan,
|
|
and only its entry list is allowed to be empty."""
|
|
plan_path = empty_proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
del written["source_sha256"]
|
|
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
|
|
|
|
code, verdict = adjudicate_empty(tmp_path, plan_path)
|
|
|
|
assert code == 2
|
|
assert "source_sha256" in capsys.readouterr().err
|
|
assert not verdict.exists()
|
|
|
|
|
|
def test_an_entries_value_that_is_not_a_list_is_still_refused(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
"""Empty is a judgement; the wrong TYPE is a malformed plan, and the two
|
|
must not collapse. The malformed one still meets the unchanged grammar.
|
|
|
|
It reaches the caller as EXIT 2, the same code every other malformed plan
|
|
already got. A grammar refusal used to escape as a traceback and exit 1,
|
|
which said "this command crashed" where the truth was "this file is not a
|
|
plan" -- and exit codes are the interface callers script against."""
|
|
plan_path = empty_proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
written["entries"] = "none"
|
|
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
|
|
|
|
code, _ = adjudicate_empty(tmp_path, plan_path)
|
|
|
|
assert code == 2
|
|
assert "segmentation_plan_invalid" in capsys.readouterr().err
|
|
assert not (tmp_path / "verdict.json").exists()
|
|
|
|
|
|
def test_a_verdict_this_tool_cannot_read_back_is_not_blamed_on_the_plan(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
"""The plan is parsed BEFORE the verdict is built, and the verdict is parsed
|
|
after. Only the first failure is the plan's.
|
|
|
|
A verdict that will not parse back is THIS command failing on what it was
|
|
told to stamp -- here an empty `--adjudicator`. Reporting that as a
|
|
malformed plan sends the operator to fix the one artifact that was fine,
|
|
which is worse than the traceback it replaced: a clean, confident, wrong
|
|
diagnosis."""
|
|
plan_path = proposal(tmp_path)
|
|
capsys.readouterr() # the proposer's own report is not what is under test
|
|
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(plan_path),
|
|
"--out",
|
|
str(tmp_path / "verdict.json"),
|
|
"--adjudicator",
|
|
"",
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
|
|
assert code == 2
|
|
stderr = capsys.readouterr().err
|
|
assert "malformed plan" not in stderr
|
|
assert "verdict" in stderr
|
|
assert not (tmp_path / "verdict.json").exists()
|
|
|
|
|
|
def test_a_malformed_non_empty_plan_exits_two_with_one_stderr_line(
|
|
tmp_path: Path, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
"""The same refusal on the other branch: a plan with entries is parsed
|
|
BEFORE anything is written, and that parse failing is a malformed plan too.
|
|
|
|
One line, because a caller reading stderr to tell malformed from missing
|
|
should not have to parse a traceback to do it."""
|
|
plan_path = proposal(tmp_path)
|
|
written = payload(plan_path)
|
|
written["entries"][0]["span"] = [10, 3]
|
|
plan_path.write_text(json.dumps(written), encoding="utf-8", newline="")
|
|
capsys.readouterr() # the proposer's own report is not what is under test
|
|
|
|
code = okf_adjudicate.main(
|
|
[
|
|
"--plan",
|
|
str(plan_path),
|
|
"--out",
|
|
str(tmp_path / "verdict.json"),
|
|
"--adjudicator",
|
|
ADJUDICATOR,
|
|
"--adjudicated-at",
|
|
AT,
|
|
]
|
|
)
|
|
|
|
assert code == 2
|
|
stderr = capsys.readouterr().err
|
|
assert len(stderr.strip().splitlines()) == 1
|
|
assert "segmentation_span_invalid" in stderr
|
|
assert not (tmp_path / "verdict.json").exists()
|