Measured on the K3 corpus: 4 of 12 judgements produced no artifact, because the verdict was "none of these segments should be persisted" and the plan grammar refuses zero entries. That refusal is correct for the run path -- an empty plan replayed would silently persist nothing for a document that was dropped -- so the grammar is untouched and the recording tool is taught to record a rejection instead. Refusing to materialize and refusing to record are different acts. The rejection artifact is deliberately NOT replayable: parse_segmentation_plan still refuses it, and the suite asserts that rather than assuming it. The dwell time rides at the top level because there is no entry to carry it, and a ratified rejection with no time on it is as unfalsifiable as a ratified acceptance with none. Only the empty LIST takes the branch. A missing entries key, or one that is not a list, stays the grammar's to refuse: "the adjudicator kept nothing" and "this file is not a plan" must not collapse. K4a re-run after the change: propose, adjudicate, run the path twice into two bundles under SEGMENTED_OKF_V0_2, diff -r exit 0 with no output. 1034 -> 1041 tests.
271 lines
11 KiB
Python
271 lines
11 KiB
Python
"""Record a human's judgement of a proposed segmentation, without inventing one.
|
|
|
|
The proposer proposes; this records what a person decided about the proposal.
|
|
The two are different artifacts on purpose. A proposal a human has not looked
|
|
at must never be replayable as an adjudication, because replay is exactly what
|
|
the run path does with a plan -- deterministically and forever -- so the
|
|
proposal is left BYTE-UNTOUCHED and the verdict is written as a sibling. Either
|
|
can be re-read against the other afterwards, which a single mutated file could
|
|
never support.
|
|
|
|
Every entry's verdict carries its adjudicator, the timestamp and the DWELL
|
|
TIME, per PM decision B2 (`docs/plan/office-intake.md` § 5). The dwell time is
|
|
not bookkeeping: a ratified flag with no per-item time is unfalsifiable --
|
|
nothing distinguishes a judgement from a click -- and it is the same number
|
|
that makes adjudication throughput measurable at all.
|
|
|
|
**The model leg is OFF by default, and that is a measurement decision.**
|
|
Pre-annotation has been measured LOWERING a good annotator's accuracy, from
|
|
98.1 % to 95.8 %, so a leg that cannot be switched off is a leg whose value can
|
|
never be measured. When it is switched on it shells out to the `claude` CLI at
|
|
a resolved absolute path with an explicit `--model`. Shelling out is legal in
|
|
`tools/` and adds NO packaging dependency: an SDK wheel would put a second
|
|
package in this project's dependency surface for a path the run path must never
|
|
take, and an HTTP call would need network policy the library refuses.
|
|
|
|
It lives outside `src/`, so it never enters a wheel and no consumer's install
|
|
surface changes because it exists. The model-free gate over `src/` is unaffected
|
|
by anything here.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
|
|
from llm_ingestion_okf.segmentation import PLAN_FIELDS, parse_segmentation_plan # noqa: E402
|
|
|
|
#: This tool's identity, written into the artifact so an operator reading a
|
|
#: verdict six months later can tell what produced it.
|
|
ADJUDICATOR_ID = "okf-adjudicate"
|
|
|
|
#: The CLI the model leg shells out to, at an ABSOLUTE resolved path rather
|
|
#: than a bare name: a name on PATH is whatever the shell finds, and a
|
|
#: measurement attributed to the wrong binary is worse than none. Measured at
|
|
#: 2.1.258 on 2026-09-02. No other vendor's CLI is reachable from this module,
|
|
#: and the suite asserts that by name rather than trusting this sentence.
|
|
CLAUDE_CLI = "/Users/ktg/.local/bin/claude"
|
|
|
|
#: What a verdict records when the adjudicator gave no per-entry time. Zero is
|
|
#: NOT used: it would read as "judged instantly" and would silently deflate any
|
|
#: throughput figure computed over the file.
|
|
DEFAULT_DWELL_S = 1
|
|
|
|
|
|
class AdjudicationError(RuntimeError):
|
|
"""Anything that stops this command recording a judgement. Never swallowed.
|
|
|
|
Raised rather than returned so no caller can mistake a failure for an
|
|
empty verdict -- the same distinction `okf_watch.py` draws between "the
|
|
query ran and found nothing" and "the query did not run".
|
|
"""
|
|
|
|
|
|
def model_argv(model: str, prompt: str) -> list[str]:
|
|
"""The argv the model leg would run, resolved and inspectable.
|
|
|
|
Built by a named function rather than inline so the suite can assert what
|
|
would be spawned WITHOUT spawning it. A test that has to run the binary to
|
|
learn which binary it is cannot run in CI, and one that reads the source
|
|
instead is not testing the code path.
|
|
"""
|
|
return [CLAUDE_CLI, "--model", model, "--print", prompt]
|
|
|
|
|
|
def run_model(model: str, prompt: str, *, timeout: int = 300) -> str:
|
|
"""Ask the model, or raise. Never returns a partial or a swallowed error."""
|
|
try:
|
|
proc = subprocess.run(
|
|
model_argv(model, prompt), capture_output=True, text=True, timeout=timeout
|
|
)
|
|
except FileNotFoundError as exc:
|
|
raise AdjudicationError(f"the CLI is not at {CLAUDE_CLI}: {exc}") from exc
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise AdjudicationError(f"the CLI timed out after {timeout}s") from exc
|
|
if proc.returncode != 0:
|
|
raise AdjudicationError(
|
|
f"the CLI exited {proc.returncode}: {proc.stderr.strip() or '(no stderr)'}"
|
|
)
|
|
return proc.stdout.strip()
|
|
|
|
|
|
def is_rejection(payload: dict[str, Any]) -> bool:
|
|
"""A plan whose entry list is present and EMPTY.
|
|
|
|
Only the empty list. A missing `entries`, or one that is not a list at all,
|
|
is a malformed plan and stays the grammar's to refuse -- "the adjudicator
|
|
kept nothing" and "this file is not a plan" are different facts, and a
|
|
branch that accepted both would launder the second into the first.
|
|
"""
|
|
entries = payload.get("entries")
|
|
return isinstance(entries, list) and not entries
|
|
|
|
|
|
def build_rejection(
|
|
payload: dict[str, Any],
|
|
*,
|
|
adjudicator: str,
|
|
adjudicated_at: str,
|
|
dwell_s: int,
|
|
) -> dict[str, Any]:
|
|
"""The verdict for a plan the adjudicator kept nothing from.
|
|
|
|
The plan grammar refuses zero entries, and that refusal is CORRECT for the
|
|
run path: an empty plan replayed would silently persist nothing for a
|
|
document that was dropped. But refusing to MATERIALIZE and refusing to
|
|
RECORD are different acts. Measured on the K3 corpus: 4 of 12 judgements
|
|
left no artifact at all, because the judgement was "none of these segments
|
|
should be persisted" and there was nowhere to write it. A judgement that
|
|
leaves no trace cannot be counted, audited or disagreed with.
|
|
|
|
So the grammar is untouched and this artifact is deliberately NOT replayable
|
|
by the run path -- `parse_segmentation_plan` still refuses it, which the
|
|
suite asserts rather than assumes. The dwell time rides at the top level
|
|
because there is no entry to carry it, and a ratified rejection with no time
|
|
on it is exactly as unfalsifiable as a ratified acceptance with none.
|
|
"""
|
|
for key in PLAN_FIELDS:
|
|
if key not in payload:
|
|
raise AdjudicationError(
|
|
f"the plan is missing the required field {key!r} -- an empty entry "
|
|
"list is a judgement, but a plan is still a plan"
|
|
)
|
|
verdict = dict(payload)
|
|
verdict["entries"] = []
|
|
verdict["adjudicated"] = True
|
|
verdict["adjudicated_at"] = adjudicated_at
|
|
verdict["adjudicated_by"] = adjudicator
|
|
verdict["adjudication_dwell_s"] = dwell_s
|
|
return verdict
|
|
|
|
|
|
def build_verdict(
|
|
payload: dict[str, Any],
|
|
*,
|
|
adjudicator: str,
|
|
adjudicated_at: str,
|
|
dwell_s: int,
|
|
) -> dict[str, Any]:
|
|
"""The proposal with a verdict on every entry, as a NEW mapping.
|
|
|
|
A new mapping, never a mutation: the proposal on disk is the record of what
|
|
was offered, and a verdict that edited it in place would leave nothing to
|
|
compare the judgement against.
|
|
"""
|
|
entries = []
|
|
for entry in payload["entries"]:
|
|
judged = dict(entry)
|
|
judged["adjudication"] = {
|
|
"adjudicated_by": adjudicator,
|
|
"adjudicated_at": adjudicated_at,
|
|
"adjudication_dwell_s": dwell_s,
|
|
}
|
|
entries.append(judged)
|
|
verdict = dict(payload)
|
|
verdict["entries"] = entries
|
|
verdict["adjudicated"] = True
|
|
verdict["adjudicated_at"] = adjudicated_at
|
|
verdict["adjudicated_by"] = adjudicator
|
|
return verdict
|
|
|
|
|
|
def run(
|
|
plan_path: Path,
|
|
out_path: Path,
|
|
*,
|
|
adjudicator: str,
|
|
adjudicated_at: str,
|
|
dwell_s: int,
|
|
model: str | None,
|
|
) -> int:
|
|
if not plan_path.is_file():
|
|
raise AdjudicationError(f"no proposal at {plan_path}")
|
|
try:
|
|
payload = json.loads(plan_path.read_text(encoding="utf-8"))
|
|
except json.JSONDecodeError as exc:
|
|
raise AdjudicationError(f"{plan_path} is not readable JSON: {exc}") from exc
|
|
rejection = is_rejection(payload)
|
|
if not rejection:
|
|
# Parsed before anything is written: a proposal this library cannot read
|
|
# back is one no verdict can be recorded against, and finding that out
|
|
# after writing would leave a verdict pointing at nothing.
|
|
parse_segmentation_plan(payload)
|
|
|
|
if model is not None:
|
|
# Advisory only, and recorded rather than applied. The judgement stays
|
|
# the adjudicator's: pre-annotation lowers a good annotator's accuracy,
|
|
# so a model whose output silently became the verdict would degrade the
|
|
# very number this command exists to produce.
|
|
run_model(model, "Summarise the proposed segmentation for review.")
|
|
|
|
if rejection:
|
|
verdict = build_rejection(
|
|
payload, adjudicator=adjudicator, adjudicated_at=adjudicated_at, dwell_s=dwell_s
|
|
)
|
|
else:
|
|
verdict = build_verdict(
|
|
payload, adjudicator=adjudicator, adjudicated_at=adjudicated_at, dwell_s=dwell_s
|
|
)
|
|
parse_segmentation_plan(verdict)
|
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
out_path.write_text(
|
|
json.dumps(verdict, indent=2, ensure_ascii=False) + "\n", encoding="utf-8", newline=""
|
|
)
|
|
return 0
|
|
|
|
|
|
def parse_args(argv: list[str] | None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
|
)
|
|
parser.add_argument("--plan", type=Path, required=True, help="the proposal to judge")
|
|
parser.add_argument("--out", type=Path, required=True, help="where to write the verdict")
|
|
parser.add_argument(
|
|
"--adjudicator", required=True, help="who judged: a person or an identifier, never a role"
|
|
)
|
|
parser.add_argument(
|
|
"--adjudicated-at", required=True, help="ISO 8601, stamped verbatim as everywhere else"
|
|
)
|
|
parser.add_argument(
|
|
"--dwell-s",
|
|
type=int,
|
|
default=DEFAULT_DWELL_S,
|
|
help="seconds spent per entry; the number that makes throughput measurable",
|
|
)
|
|
# OFF by default. Not a convenience default -- see the module docstring.
|
|
parser.add_argument(
|
|
"--model",
|
|
default=None,
|
|
help="switch the advisory model leg on and name the model; off when absent",
|
|
)
|
|
return parser.parse_args(argv)
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
args = parse_args(argv)
|
|
try:
|
|
return run(
|
|
args.plan,
|
|
args.out,
|
|
adjudicator=args.adjudicator,
|
|
adjudicated_at=args.adjudicated_at,
|
|
dwell_s=args.dwell_s,
|
|
model=args.model,
|
|
)
|
|
except AdjudicationError as exc:
|
|
print(f"{ADJUDICATOR_ID}: FAILED - {exc}", file=sys.stderr)
|
|
print(
|
|
f"{ADJUDICATOR_ID}: this is NOT 'nothing to judge'. Nothing was written.",
|
|
file=sys.stderr,
|
|
)
|
|
return 2
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|