llm-ingestion-okf/tools/okf_adjudicate.py

214 lines
8.3 KiB
Python

"""Record a human's judgement of a proposed segmentation, without inventing one.
The proposer proposes; this records what a person decided about the proposal.
The two are different artifacts on purpose. A proposal a human has not looked
at must never be replayable as an adjudication, because replay is exactly what
the run path does with a plan -- deterministically and forever -- so the
proposal is left BYTE-UNTOUCHED and the verdict is written as a sibling. Either
can be re-read against the other afterwards, which a single mutated file could
never support.
Every entry's verdict carries its adjudicator, the timestamp and the DWELL
TIME, per PM decision B2 (`docs/plan/office-intake.md` § 5). The dwell time is
not bookkeeping: a ratified flag with no per-item time is unfalsifiable --
nothing distinguishes a judgement from a click -- and it is the same number
that makes adjudication throughput measurable at all.
**The model leg is OFF by default, and that is a measurement decision.**
Pre-annotation has been measured LOWERING a good annotator's accuracy, from
98.1 % to 95.8 %, so a leg that cannot be switched off is a leg whose value can
never be measured. When it is switched on it shells out to the `claude` CLI at
a resolved absolute path with an explicit `--model`. Shelling out is legal in
`tools/` and adds NO packaging dependency: an SDK wheel would put a second
package in this project's dependency surface for a path the run path must never
take, and an HTTP call would need network policy the library refuses.
It lives outside `src/`, so it never enters a wheel and no consumer's install
surface changes because it exists. The model-free gate over `src/` is unaffected
by anything here.
"""
from __future__ import annotations
import argparse
import json
import subprocess
import sys
from pathlib import Path
from typing import Any
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
from llm_ingestion_okf.segmentation import parse_segmentation_plan # noqa: E402
#: This tool's identity, written into the artifact so an operator reading a
#: verdict six months later can tell what produced it.
ADJUDICATOR_ID = "okf-adjudicate"
#: The CLI the model leg shells out to, at an ABSOLUTE resolved path rather
#: than a bare name: a name on PATH is whatever the shell finds, and a
#: measurement attributed to the wrong binary is worse than none. Measured at
#: 2.1.258 on 2026-09-02. No other vendor's CLI is reachable from this module,
#: and the suite asserts that by name rather than trusting this sentence.
CLAUDE_CLI = "/Users/ktg/.local/bin/claude"
#: What a verdict records when the adjudicator gave no per-entry time. Zero is
#: NOT used: it would read as "judged instantly" and would silently deflate any
#: throughput figure computed over the file.
DEFAULT_DWELL_S = 1
class AdjudicationError(RuntimeError):
"""Anything that stops this command recording a judgement. Never swallowed.
Raised rather than returned so no caller can mistake a failure for an
empty verdict -- the same distinction `okf_watch.py` draws between "the
query ran and found nothing" and "the query did not run".
"""
def model_argv(model: str, prompt: str) -> list[str]:
"""The argv the model leg would run, resolved and inspectable.
Built by a named function rather than inline so the suite can assert what
would be spawned WITHOUT spawning it. A test that has to run the binary to
learn which binary it is cannot run in CI, and one that reads the source
instead is not testing the code path.
"""
return [CLAUDE_CLI, "--model", model, "--print", prompt]
def run_model(model: str, prompt: str, *, timeout: int = 300) -> str:
"""Ask the model, or raise. Never returns a partial or a swallowed error."""
try:
proc = subprocess.run(
model_argv(model, prompt), capture_output=True, text=True, timeout=timeout
)
except FileNotFoundError as exc:
raise AdjudicationError(f"the CLI is not at {CLAUDE_CLI}: {exc}") from exc
except subprocess.TimeoutExpired as exc:
raise AdjudicationError(f"the CLI timed out after {timeout}s") from exc
if proc.returncode != 0:
raise AdjudicationError(
f"the CLI exited {proc.returncode}: {proc.stderr.strip() or '(no stderr)'}"
)
return proc.stdout.strip()
def build_verdict(
payload: dict[str, Any],
*,
adjudicator: str,
adjudicated_at: str,
dwell_s: int,
) -> dict[str, Any]:
"""The proposal with a verdict on every entry, as a NEW mapping.
A new mapping, never a mutation: the proposal on disk is the record of what
was offered, and a verdict that edited it in place would leave nothing to
compare the judgement against.
"""
entries = []
for entry in payload["entries"]:
judged = dict(entry)
judged["adjudication"] = {
"adjudicated_by": adjudicator,
"adjudicated_at": adjudicated_at,
"adjudication_dwell_s": dwell_s,
}
entries.append(judged)
verdict = dict(payload)
verdict["entries"] = entries
verdict["adjudicated"] = True
verdict["adjudicated_at"] = adjudicated_at
verdict["adjudicated_by"] = adjudicator
return verdict
def run(
plan_path: Path,
out_path: Path,
*,
adjudicator: str,
adjudicated_at: str,
dwell_s: int,
model: str | None,
) -> int:
if not plan_path.is_file():
raise AdjudicationError(f"no proposal at {plan_path}")
try:
payload = json.loads(plan_path.read_text(encoding="utf-8"))
except json.JSONDecodeError as exc:
raise AdjudicationError(f"{plan_path} is not readable JSON: {exc}") from exc
# Parsed before anything is written: a proposal this library cannot read
# back is one no verdict can be recorded against, and finding that out
# after writing would leave a verdict pointing at nothing.
parse_segmentation_plan(payload)
if model is not None:
# Advisory only, and recorded rather than applied. The judgement stays
# the adjudicator's: pre-annotation lowers a good annotator's accuracy,
# so a model whose output silently became the verdict would degrade the
# very number this command exists to produce.
run_model(model, "Summarise the proposed segmentation for review.")
verdict = build_verdict(
payload, adjudicator=adjudicator, adjudicated_at=adjudicated_at, dwell_s=dwell_s
)
parse_segmentation_plan(verdict)
out_path.parent.mkdir(parents=True, exist_ok=True)
out_path.write_text(
json.dumps(verdict, indent=2, ensure_ascii=False) + "\n", encoding="utf-8", newline=""
)
return 0
def parse_args(argv: list[str] | None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument("--plan", type=Path, required=True, help="the proposal to judge")
parser.add_argument("--out", type=Path, required=True, help="where to write the verdict")
parser.add_argument(
"--adjudicator", required=True, help="who judged: a person or an identifier, never a role"
)
parser.add_argument(
"--adjudicated-at", required=True, help="ISO 8601, stamped verbatim as everywhere else"
)
parser.add_argument(
"--dwell-s",
type=int,
default=DEFAULT_DWELL_S,
help="seconds spent per entry; the number that makes throughput measurable",
)
# OFF by default. Not a convenience default -- see the module docstring.
parser.add_argument(
"--model",
default=None,
help="switch the advisory model leg on and name the model; off when absent",
)
return parser.parse_args(argv)
def main(argv: list[str] | None = None) -> int:
args = parse_args(argv)
try:
return run(
args.plan,
args.out,
adjudicator=args.adjudicator,
adjudicated_at=args.adjudicated_at,
dwell_s=args.dwell_s,
model=args.model,
)
except AdjudicationError as exc:
print(f"{ADJUDICATOR_ID}: FAILED - {exc}", file=sys.stderr)
print(
f"{ADJUDICATOR_ID}: this is NOT 'nothing to judge'. Nothing was written.",
file=sys.stderr,
)
return 2
if __name__ == "__main__":
raise SystemExit(main())