run() parses the plan before building anything and parses its own
verdict after. Only the first failure is the operator's file.
The second is reachable: an empty --adjudicator produces a verdict the
grammar refuses ("adjudication field 'adjudicated_by' must be a
non-empty string"). The plan parsed fine; the fault is in what this
command stamped onto it.
Catching SegmentationError at the top of main() -- the previous commit
-- caught both raise sites and printed "malformed plan" for each. On
this path that is a clean, confident, WRONG diagnosis: it sends the
operator to fix the one artifact that was fine. Worse than the traceback
it replaced, because a traceback at least does not claim to know.
The verdict parse now raises AdjudicationError, which is what "this
command failed" already means in this file and already returns 2. Exit
code unchanged either way, nothing written either way; only the message
changes.
Suite 1073 -> 1074 passed (pytest exit 0, measured without a pipe);
ruff and mypy --strict clean.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
289 lines
12 KiB
Python
289 lines
12 KiB
Python
"""Record a human's judgement of a proposed segmentation, without inventing one.
|
|
|
|
The proposer proposes; this records what a person decided about the proposal.
|
|
The two are different artifacts on purpose. A proposal a human has not looked
|
|
at must never be replayable as an adjudication, because replay is exactly what
|
|
the run path does with a plan -- deterministically and forever -- so the
|
|
proposal is left BYTE-UNTOUCHED and the verdict is written as a sibling. Either
|
|
can be re-read against the other afterwards, which a single mutated file could
|
|
never support.
|
|
|
|
Every entry's verdict carries its adjudicator, the timestamp and the DWELL
|
|
TIME, per PM decision B2 (`docs/plan/office-intake.md` § 5). The dwell time is
|
|
not bookkeeping: a ratified flag with no per-item time is unfalsifiable --
|
|
nothing distinguishes a judgement from a click -- and it is the same number
|
|
that makes adjudication throughput measurable at all.
|
|
|
|
**The model leg is OFF by default, and that is a measurement decision.**
|
|
Pre-annotation has been measured LOWERING a good annotator's accuracy, from
|
|
98.1 % to 95.8 %, so a leg that cannot be switched off is a leg whose value can
|
|
never be measured. When it is switched on it shells out to the `claude` CLI at
|
|
a resolved absolute path with an explicit `--model`. Shelling out is legal in
|
|
`tools/` and adds NO packaging dependency: an SDK wheel would put a second
|
|
package in this project's dependency surface for a path the run path must never
|
|
take, and an HTTP call would need network policy the library refuses.
|
|
|
|
It lives outside `src/`, so it never enters a wheel and no consumer's install
|
|
surface changes because it exists. The model-free gate over `src/` is unaffected
|
|
by anything here.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
|
|
from llm_ingestion_okf.errors import SegmentationError # noqa: E402
|
|
from llm_ingestion_okf.segmentation import PLAN_FIELDS, parse_segmentation_plan # noqa: E402
|
|
|
|
#: This tool's identity, written into the artifact so an operator reading a
|
|
#: verdict six months later can tell what produced it.
|
|
ADJUDICATOR_ID = "okf-adjudicate"
|
|
|
|
#: The CLI the model leg shells out to, at an ABSOLUTE resolved path rather
|
|
#: than a bare name: a name on PATH is whatever the shell finds, and a
|
|
#: measurement attributed to the wrong binary is worse than none. Measured at
|
|
#: 2.1.258 on 2026-09-02. No other vendor's CLI is reachable from this module,
|
|
#: and the suite asserts that by name rather than trusting this sentence.
|
|
CLAUDE_CLI = "/Users/ktg/.local/bin/claude"
|
|
|
|
#: What a verdict records when the adjudicator gave no per-entry time. Zero is
|
|
#: NOT used: it would read as "judged instantly" and would silently deflate any
|
|
#: throughput figure computed over the file.
|
|
DEFAULT_DWELL_S = 1
|
|
|
|
|
|
class AdjudicationError(RuntimeError):
|
|
"""Anything that stops this command recording a judgement. Never swallowed.
|
|
|
|
Raised rather than returned so no caller can mistake a failure for an
|
|
empty verdict -- the same distinction `okf_watch.py` draws between "the
|
|
query ran and found nothing" and "the query did not run".
|
|
"""
|
|
|
|
|
|
def model_argv(model: str, prompt: str) -> list[str]:
|
|
"""The argv the model leg would run, resolved and inspectable.
|
|
|
|
Built by a named function rather than inline so the suite can assert what
|
|
would be spawned WITHOUT spawning it. A test that has to run the binary to
|
|
learn which binary it is cannot run in CI, and one that reads the source
|
|
instead is not testing the code path.
|
|
"""
|
|
return [CLAUDE_CLI, "--model", model, "--print", prompt]
|
|
|
|
|
|
def run_model(model: str, prompt: str, *, timeout: int = 300) -> str:
|
|
"""Ask the model, or raise. Never returns a partial or a swallowed error."""
|
|
try:
|
|
proc = subprocess.run(
|
|
model_argv(model, prompt), capture_output=True, text=True, timeout=timeout
|
|
)
|
|
except FileNotFoundError as exc:
|
|
raise AdjudicationError(f"the CLI is not at {CLAUDE_CLI}: {exc}") from exc
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise AdjudicationError(f"the CLI timed out after {timeout}s") from exc
|
|
if proc.returncode != 0:
|
|
raise AdjudicationError(
|
|
f"the CLI exited {proc.returncode}: {proc.stderr.strip() or '(no stderr)'}"
|
|
)
|
|
return proc.stdout.strip()
|
|
|
|
|
|
def is_rejection(payload: dict[str, Any]) -> bool:
|
|
"""A plan whose entry list is present and EMPTY.
|
|
|
|
Only the empty list. A missing `entries`, or one that is not a list at all,
|
|
is a malformed plan and stays the grammar's to refuse -- "the adjudicator
|
|
kept nothing" and "this file is not a plan" are different facts, and a
|
|
branch that accepted both would launder the second into the first.
|
|
"""
|
|
entries = payload.get("entries")
|
|
return isinstance(entries, list) and not entries
|
|
|
|
|
|
def build_rejection(
|
|
payload: dict[str, Any],
|
|
*,
|
|
adjudicator: str,
|
|
adjudicated_at: str,
|
|
dwell_s: int,
|
|
) -> dict[str, Any]:
|
|
"""The verdict for a plan the adjudicator kept nothing from.
|
|
|
|
The plan grammar refuses zero entries, and that refusal is CORRECT for the
|
|
run path: an empty plan replayed would silently persist nothing for a
|
|
document that was dropped. But refusing to MATERIALIZE and refusing to
|
|
RECORD are different acts. Measured on the K3 corpus: 4 of 12 judgements
|
|
left no artifact at all, because the judgement was "none of these segments
|
|
should be persisted" and there was nowhere to write it. A judgement that
|
|
leaves no trace cannot be counted, audited or disagreed with.
|
|
|
|
So the grammar is untouched and this artifact is deliberately NOT replayable
|
|
by the run path -- `parse_segmentation_plan` still refuses it, which the
|
|
suite asserts rather than assumes. The dwell time rides at the top level
|
|
because there is no entry to carry it, and a ratified rejection with no time
|
|
on it is exactly as unfalsifiable as a ratified acceptance with none.
|
|
"""
|
|
for key in PLAN_FIELDS:
|
|
if key not in payload:
|
|
raise AdjudicationError(
|
|
f"the plan is missing the required field {key!r} -- an empty entry "
|
|
"list is a judgement, but a plan is still a plan"
|
|
)
|
|
verdict = dict(payload)
|
|
verdict["entries"] = []
|
|
verdict["adjudicated"] = True
|
|
verdict["adjudicated_at"] = adjudicated_at
|
|
verdict["adjudicated_by"] = adjudicator
|
|
verdict["adjudication_dwell_s"] = dwell_s
|
|
return verdict
|
|
|
|
|
|
def build_verdict(
|
|
payload: dict[str, Any],
|
|
*,
|
|
adjudicator: str,
|
|
adjudicated_at: str,
|
|
dwell_s: int,
|
|
) -> dict[str, Any]:
|
|
"""The proposal with a verdict on every entry, as a NEW mapping.
|
|
|
|
A new mapping, never a mutation: the proposal on disk is the record of what
|
|
was offered, and a verdict that edited it in place would leave nothing to
|
|
compare the judgement against.
|
|
"""
|
|
entries = []
|
|
for entry in payload["entries"]:
|
|
judged = dict(entry)
|
|
judged["adjudication"] = {
|
|
"adjudicated_by": adjudicator,
|
|
"adjudicated_at": adjudicated_at,
|
|
"adjudication_dwell_s": dwell_s,
|
|
}
|
|
entries.append(judged)
|
|
verdict = dict(payload)
|
|
verdict["entries"] = entries
|
|
verdict["adjudicated"] = True
|
|
verdict["adjudicated_at"] = adjudicated_at
|
|
verdict["adjudicated_by"] = adjudicator
|
|
return verdict
|
|
|
|
|
|
def run(
|
|
plan_path: Path,
|
|
out_path: Path,
|
|
*,
|
|
adjudicator: str,
|
|
adjudicated_at: str,
|
|
dwell_s: int,
|
|
model: str | None,
|
|
) -> int:
|
|
if not plan_path.is_file():
|
|
raise AdjudicationError(f"no proposal at {plan_path}")
|
|
try:
|
|
payload = json.loads(plan_path.read_text(encoding="utf-8"))
|
|
except json.JSONDecodeError as exc:
|
|
raise AdjudicationError(f"{plan_path} is not readable JSON: {exc}") from exc
|
|
rejection = is_rejection(payload)
|
|
if not rejection:
|
|
# Parsed before anything is written: a proposal this library cannot read
|
|
# back is one no verdict can be recorded against, and finding that out
|
|
# after writing would leave a verdict pointing at nothing.
|
|
parse_segmentation_plan(payload)
|
|
|
|
if model is not None:
|
|
# Advisory only, and recorded rather than applied. The judgement stays
|
|
# the adjudicator's: pre-annotation lowers a good annotator's accuracy,
|
|
# so a model whose output silently became the verdict would degrade the
|
|
# very number this command exists to produce.
|
|
run_model(model, "Summarise the proposed segmentation for review.")
|
|
|
|
if rejection:
|
|
verdict = build_rejection(
|
|
payload, adjudicator=adjudicator, adjudicated_at=adjudicated_at, dwell_s=dwell_s
|
|
)
|
|
else:
|
|
verdict = build_verdict(
|
|
payload, adjudicator=adjudicator, adjudicated_at=adjudicated_at, dwell_s=dwell_s
|
|
)
|
|
# The plan already parsed above, so a failure HERE is this command's
|
|
# own output, not the operator's file. Raised as an AdjudicationError
|
|
# so it is not reported as a malformed plan: that would send the
|
|
# operator to fix the one artifact that was fine.
|
|
try:
|
|
parse_segmentation_plan(verdict)
|
|
except SegmentationError as exc:
|
|
raise AdjudicationError(
|
|
f"the verdict this command built does not parse back [{exc.code}]: {exc}"
|
|
) from exc
|
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
out_path.write_text(
|
|
json.dumps(verdict, indent=2, ensure_ascii=False) + "\n", encoding="utf-8", newline=""
|
|
)
|
|
return 0
|
|
|
|
|
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
|
)
|
|
parser.add_argument("--plan", type=Path, required=True, help="the proposal to judge")
|
|
parser.add_argument("--out", type=Path, required=True, help="where to write the verdict")
|
|
parser.add_argument(
|
|
"--adjudicator", required=True, help="who judged: a person or an identifier, never a role"
|
|
)
|
|
parser.add_argument(
|
|
"--adjudicated-at", required=True, help="ISO 8601, stamped verbatim as everywhere else"
|
|
)
|
|
parser.add_argument(
|
|
"--dwell-s",
|
|
type=int,
|
|
default=DEFAULT_DWELL_S,
|
|
help="seconds spent per entry; the number that makes throughput measurable",
|
|
)
|
|
# OFF by default. Not a convenience default -- see the module docstring.
|
|
parser.add_argument(
|
|
"--model",
|
|
default=None,
|
|
help="switch the advisory model leg on and name the model; off when absent",
|
|
)
|
|
return parser.parse_args(argv)
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
args = parse_args(argv)
|
|
try:
|
|
return run(
|
|
args.plan,
|
|
args.out,
|
|
adjudicator=args.adjudicator,
|
|
adjudicated_at=args.adjudicated_at,
|
|
dwell_s=args.dwell_s,
|
|
model=args.model,
|
|
)
|
|
except AdjudicationError as exc:
|
|
print(f"{ADJUDICATOR_ID}: FAILED - {exc}", file=sys.stderr)
|
|
print(
|
|
f"{ADJUDICATOR_ID}: this is NOT 'nothing to judge'. Nothing was written.",
|
|
file=sys.stderr,
|
|
)
|
|
return 2
|
|
except SegmentationError as exc:
|
|
# A malformed plan is a refusal, not a crash. Letting the grammar's
|
|
# error escape gave a traceback and exit 1, which a caller scripting on
|
|
# exit codes reads as "this command broke" -- the same code an unhandled
|
|
# bug would produce. One line, and the same 2 every other malformed plan
|
|
# already got.
|
|
print(f"{ADJUDICATOR_ID}: FAILED - malformed plan [{exc.code}]: {exc}", file=sys.stderr)
|
|
return 2
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|