test(soek): a red search gate measuring what the asker RECEIVES, one exit code

`tools/okf_soek_gate.py` runs the question sets through the same
`consume.build_payload` path `okf consume` and MCP's `okf_ask` use, at the
shipped k and limit, and counts on the DELIVERED excerpts rather than on an
internal ranking -- a rank nobody receives is not an answer.

Written RED: it is the finish line a later round has to make green, and a
gate that is green on the day it is written has measured nothing. Thresholds
are named constants carrying their reason in a comment. Exit 0 only when every
row holds, 1 with the full table otherwise, 2 when the collection cannot be
found -- never 0 hits against nothing. Two runs print identical bytes.

The sets and the collection are both INPUTS: nothing about anyone's corpus is
committed, and no row names a document, a question or a quote.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Kjell Tore Guttormsen 2026-09-21 04:03:11 +02:00
commit 245ff64c3d
2 changed files with 1051 additions and 0 deletions

767
tools/okf_soek_gate.py Normal file
View file

@ -0,0 +1,767 @@
"""The search gate for `okf consume` -- one command, one exit code (order B).
WHAT IT ASKS. For a collection and a frozen question set: of N measurement
units, how many does the payload a reader actually RECEIVES carry the fasit
for? It measures the DELIVERY at the shipped defaults (`consume.DEFAULT_K`,
`consume.DEFAULT_LIMIT`), never an internal rank -- a concept the ranker found
and the cut dropped is a miss here, because it is a miss for the person asking.
WRITTEN RED, BEFORE ANY CAPABILITY. Nothing in this module changes the
ranking, the fusion, the tokenisation, the cut, the segmentation or the
defaults; it only measures them. It goes through `consume.build_payload`, the
one entry point `okf consume` and the MCP server's `okf_ask` both use, so a
number here is a number about the shipped product and not about a harness.
THE GATE IS NOT IN THE TEST SUITE. It is red against a real collection by
construction, and a red test in a green suite is a suite nobody reads. The
measuring instrument -- the hit rule, the counting and the missing-fixture
state -- IS in the suite, against a synthetic corpus
(`tests/test_soek_gate.py`).
THE SETS ARE INPUTS, NEVER CONSTANTS. `tools/okf_retrieval_gate.py` states the
rule and this module inherits it: a real gold set names documents in a
consumer's corpus, and this repository is public. A set arrives as a file under
`--sets` (default `eval/soek/`). A set that is ABSENT is reported
`IKKE KJOERT -- fixture mangler` and counts RED: "not run" and "no hits" are
two different facts about the world, and collapsing them would let a gate go
green by having less to measure.
THE HIT RULE IS THE SETS' OWN, VERBATIM. From the sets' `hit_rule`
field: "A question is answered with source when at least one payload excerpt
has source_file == <doc>.md for a fasit entry AND contains that entry's quote
(case-insensitive, whitespace collapsed). Any fasit entry suffices."
EVERY MISS CARRIES ONE CLASS AND NOT A GUESS. `byggefeil` -- no fasit quote is
in the collection at all, so no ranking could have delivered it. `soekefeil` --
a fasit quote IS in the collection and was not delivered. The second denominator
is read off the concept files on disk, never off the payload: the judge opens
the bundle.
THE COLUMN HEADS AND THE NOT-RUN MARKER ARE THE ORDER'S WORDS. Everything else
here is English, per this repository's convention for a public repo.
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from collections.abc import Mapping, Sequence
from dataclasses import dataclass
from pathlib import Path
TOOLS = Path(__file__).resolve().parent
REPO = TOOLS.parent
if str(REPO / "src") not in sys.path:
sys.path.insert(0, str(REPO / "src"))
if str(TOOLS) not in sys.path:
sys.path.insert(0, str(TOOLS))
from okf_retrieval_gate import marked # noqa: E402
from llm_ingestion_okf import consume # noqa: E402
#: One repository, one reading of "the payload says the bundle does not cover
#: this". `okf_retrieval_gate.marked` carries the measurement and the bar
#: (`UNANSWERED_BAR`); a second definition here would let the two gates
#: disagree about the same bytes.
uncovered_signal = marked
DEFAULT_SET_DIR = REPO / "eval" / "soek"
MISSING_FIXTURE = "IKKE KJOERT -- fixture mangler"
BUILD_FAILURE = "byggefeil"
SEARCH_FAILURE = "soekefeil"
#: PM's noise finding from the spike, reported rather than gated: table
#: fragments titled `Tabell linje N` rank high and carry nothing.
NOISE_TITLE = re.compile(r"^Tabell linje \d+$")
# --- the thresholds -----------------------------------------------------------
#
# PM's, measured in the search spike of 2026-09-20 against the same
# collection, and changed only by PM. The spike measured RANK; these rows
# measure DELIVERY, so a divergence is expected and is explained per row in the
# run report rather than absorbed by moving a bar.
#
# The floor rows are floors and not targets. `THRESHOLD_HOLDOUT` in particular
# guards against over-fitting: the hold-out set is run and reported and is
# never something anyone tunes against -- a change that lifts the phase set and
# not this one learned the answer key.
THRESHOLD_PHASE = 18 # series (a)
THRESHOLD_RELEASE_ONLY = 7 # series (b)
THRESHOLD_HOLDOUT = 6 # series (c), a FLOOR, never a target
THRESHOLD_NORWEGIAN_DIRECT = 6 # series (d), no regression
THRESHOLD_NORWEGIAN_SUBQUESTIONS = 16 # series (e)
THRESHOLD_OPERATOR = 4 # fasit places -- series (f), via `OP_kart`
THRESHOLD_NEGATIVE_FLAGGED = 4 # series (g)
THRESHOLD_POSITIVE_MISFLAGGED = 2 # at most, over the English positives
THRESHOLD_LARGEST_EXCERPT = 6_000 # characters, at most, in any delivered excerpt
class GateUsage(Exception):
"""Wrong input: exit 2, never a quiet row."""
# --- the hit rule -------------------------------------------------------------
def collapse(text: str) -> str:
"""The sets' own comparison form: case folded, whitespace collapsed."""
return " ".join(text.lower().split())
def excerpt_carries(excerpt: Mapping[str, object], doc: str, quote: str) -> bool:
"""One excerpt against one fasit entry -- BOTH halves, never either alone.
The source half alone would credit any excerpt from the right document, and
the quote half alone would credit a document that merely repeats a line the
fasit names elsewhere.
"""
if excerpt.get("source_file") != f"{doc}.md":
return False
return collapse(quote) in collapse(str(excerpt.get("text", "")))
def question_hit(
excerpts: Sequence[Mapping[str, object]], fasit: Sequence[Mapping[str, str]]
) -> bool:
"""Any fasit entry suffices -- both sets say so in their own `hit_rule`."""
return any(
excerpt_carries(excerpt, entry["doc"], entry["quote"])
for excerpt in excerpts
for entry in fasit
)
def place_delivered(excerpts: Sequence[Mapping[str, object]], place: Mapping[str, str]) -> bool:
"""One of the operator's fasit PLACES, `{doc, section}`.
The set's own `hit_rule` is prose and cannot be executed; this is the same
SHAPE as the sets' own rule (the right source AND containment), with the
section name in place of a quote, matched against the excerpt's text or its
title -- a section can be delivered as a concept whose title IS the section.
The set's declared `hit_rule` string is printed beside the row so a reader
can check this implementation against it.
"""
section = collapse(place.get("section", ""))
for excerpt in excerpts:
if excerpt.get("source_file") != f"{place['doc']}.md":
continue
if not section:
return True
if section in collapse(str(excerpt.get("text", ""))):
return True
if section in collapse(str(excerpt.get("title", ""))):
return True
return False
# --- the collection -----------------------------------------------------------
def collection_text(bundle_root: Path) -> dict[str, str]:
"""`source_file` -> the collapsed text of every concept written from it.
Read off the concept FILES, because this is the denominator that separates
a build failure from a search failure and the payload cannot answer it: a
quote the collection never held is not a ranking's fault.
"""
root_bundle_id = consume.root_bundle_id_of(bundle_root)
text: dict[str, list[str]] = {}
for concept_id in consume.enumerate_concepts(bundle_root):
concept = consume.read_concept(
consume.read_path_in_bundle(bundle_root, f"{concept_id}.md"),
bundle_root=bundle_root,
root_bundle_id=root_bundle_id,
)
source = concept.source_file or ""
text.setdefault(source, []).append(concept.body)
return {source: collapse(" ".join(bodies)) for source, bodies in text.items()}
def classify_miss(
question_id: str, fasit: Sequence[Mapping[str, str]], text: Mapping[str, str]
) -> "Miss":
"""One class per miss, never two and never none."""
present = [
entry["doc"]
for entry in fasit
if collapse(entry["quote"]) in text.get(f"{entry['doc']}.md", "")
]
if present:
return Miss(
question_id, SEARCH_FAILURE, f"in the collection ({', '.join(present)}), not delivered"
)
return Miss(question_id, BUILD_FAILURE, "no fasit quote is in the collection")
# --- the rows -----------------------------------------------------------------
@dataclass(frozen=True)
class Miss:
question_id: str
klass: str
detail: str
def render(self) -> str:
return f" {self.question_id:<10} {self.klass:<10} {self.detail}"
@dataclass
class Row:
key: str
label: str
measured: int | None
denominator: int
threshold: int
at_most: bool = False
unit: str = ""
note: str = ""
misses: tuple[Miss, ...] = ()
def holds(self) -> bool | None:
"""`None` is the third state: the row did not run, and that is red."""
if self.measured is None:
return None
if self.at_most:
return self.measured <= self.threshold
return self.measured >= self.threshold
def render(self) -> str:
bar = f"{'<=' if self.at_most else '>='} {self.threshold}"
if self.measured is None:
return f" {self.label:<44} {MISSING_FIXTURE:<22} {bar:<9} NEI"
# `NEI` and not a blank: a row nobody measured has not held.
value = f"{self.measured}{self.unit}"
if self.denominator:
value = f"{self.measured} / {self.denominator}"
verdict = "JA" if self.holds() else "NEI"
return f" {self.label:<44} {value:<22} {bar:<9} {verdict}"
@dataclass
class Report:
collection: str
rows: list[Row]
notes: Sequence[str] = ()
def row(self, key: str) -> Row:
for row in self.rows:
if row.key == key:
return row
raise KeyError(key)
def exit_code(self) -> int:
return 0 if all(row.holds() for row in self.rows) else 1
def render(self) -> str:
lines = [
"OKF SOEK-PORT -- what the asker actually RECEIVES, at the shipped defaults",
f" k = {consume.DEFAULT_K}, limit = {consume.DEFAULT_LIMIT}, "
f"contract = {consume.CONTRACT_REVISION}",
f" collection: {self.collection}",
"",
f" {'serie':<44} {'maaltall':<22} {'terskel':<9} holder",
f" {'-' * 44} {'-' * 22} {'-' * 9} ------",
]
lines.extend(row.render() for row in self.rows)
lines.append("")
for row in self.rows:
if not row.misses and not row.note:
continue
lines.append(f" {row.label}")
if row.note:
lines.append(f" note: {row.note}")
lines.extend(miss.render() for miss in row.misses)
lines.append("")
if self.notes:
lines.append(" notes")
lines.extend(f" {note}" for note in self.notes)
lines.append("")
lines.append("GATE GROENN" if self.exit_code() == 0 else "GATE ROED")
return "\n".join(lines) + "\n"
# --- the sets -----------------------------------------------------------------
@dataclass(frozen=True)
class Sets:
phase: Mapping[str, object] | None = None
holdout: Mapping[str, object] | None = None
norwegian: Mapping[str, object] | None = None
subquestions: Mapping[str, object] | None = None
SET_FILES = {
"phase": "fase-sporsmaal.json",
"holdout": "holdout-sporsmaal.json",
"norwegian": "norske-sporsmaal.json",
"subquestions": "delsporsmaal.json",
}
def load_sets(directory: Path) -> Sets:
"""Absent is a red row; unreadable is wrong input.
The two are different facts and the second must never read as the first: a
set that was placed and cannot be parsed is a mistake someone can fix now,
and swallowing it as `IKKE KJOERT` would hide it behind a row that is red
anyway.
"""
loaded: dict[str, Mapping[str, object] | None] = {}
for key, name in SET_FILES.items():
path = directory / name
if not path.is_file():
loaded[key] = None
continue
try:
loaded[key] = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError) as error:
raise GateUsage(f"{name} could not be read: {error}") from error
return Sets(**loaded)
# --- asking -------------------------------------------------------------------
class Asker:
"""`consume.build_payload`, memoised on the question.
The memo is sound because `build_payload` documents itself pure with
respect to the clock and the network: the same bundle bytes and the same
question return the same object. It exists because series (b) is a subset
of (a) and series (g)'s false-flag denominator IS the payloads (a) and
(c) already built -- re-asking them would cost minutes and could not
change an answer.
"""
def __init__(self, bundle_root: Path) -> None:
self.bundle_root = bundle_root
self._memo: dict[str, Mapping[str, object]] = {}
def __call__(self, question: str) -> Mapping[str, object]:
if question not in self._memo:
self._memo[question] = consume.build_payload(self.bundle_root, question=question)
return self._memo[question]
def many(self, questions: Sequence[str]) -> Mapping[str, object]:
"""Every sub-question in ONE call, merged by the product (v1.1 C2).
Series (e) and (f) measure the merge a reader actually receives, at the
shipped `k`; the gate carries no merge of its own.
"""
if not questions:
# A question the set gives no sub-questions for delivers nothing
# through this route; the product refuses an empty call.
return {"excerpts": []}
return consume.build_multi_payload(self.bundle_root, questions=list(questions))
def excerpts_of(payload: Mapping[str, object]) -> list[Mapping[str, object]]:
excerpts = payload.get("excerpts", [])
assert isinstance(excerpts, list)
return excerpts
# --- the series ---------------------------------------------------------------
def _english_series(
questions: Sequence[Mapping[str, object]], ask: Asker, text: Mapping[str, str]
) -> tuple[int, list[Miss], list[Mapping[str, object]]]:
hits = 0
misses: list[Miss] = []
delivered: list[Mapping[str, object]] = []
for question in questions:
payload = ask(str(question["question"]))
excerpts = excerpts_of(payload)
delivered.extend(excerpts)
fasit = question["fasit"]
assert isinstance(fasit, list)
if question_hit(excerpts, fasit):
hits += 1
else:
misses.append(classify_miss(str(question["id"]), fasit, text))
return hits, misses, delivered
def run(bundle_root: Path, sets: Sets) -> Report:
"""Every series, in the order the order names them."""
ask = Asker(bundle_root)
text = collection_text(bundle_root)
rows: list[Row] = []
delivered_everywhere: list[Mapping[str, object]] = []
english_positive_payloads: list[Mapping[str, object]] = []
positives_complete = True
# (a) and (b): one ask, two rows. `release_only` is a CLASS within the
# phase set, so asking it again would be a second measurement of the same
# payloads and could only differ by accident.
if sets.phase is None:
rows.append(Row("a", "(a) phase, hit in the delivery", None, 0, THRESHOLD_PHASE))
rows.append(Row("b", "(b) of which release_only", None, 0, THRESHOLD_RELEASE_ONLY))
positives_complete = False
else:
questions = sets.phase["questions"]
assert isinstance(questions, list)
hits, misses, delivered = _english_series(questions, ask, text)
delivered_everywhere.extend(delivered)
english_positive_payloads.extend(ask(str(q["question"])) for q in questions)
rows.append(
Row(
"a",
"(a) phase, hit in the delivery",
hits,
len(questions),
THRESHOLD_PHASE,
misses=tuple(misses),
)
)
release_only = [q for q in questions if q.get("class") == "release_only"]
release_hits, release_misses, _ = _english_series(release_only, ask, text)
rows.append(
Row(
"b",
"(b) of which release_only",
release_hits,
len(release_only),
THRESHOLD_RELEASE_ONLY,
misses=tuple(release_misses),
)
)
# (c) the hold-out set. RUN AND REPORTED, NEVER TUNED AGAINST.
if sets.holdout is None:
rows.append(Row("c", "(c) hold-out", None, 0, THRESHOLD_HOLDOUT, note=_HOLDOUT_NOTE))
positives_complete = False
else:
questions = sets.holdout["questions"]
assert isinstance(questions, list)
hits, misses, delivered = _english_series(questions, ask, text)
delivered_everywhere.extend(delivered)
english_positive_payloads.extend(ask(str(q["question"])) for q in questions)
rows.append(
Row(
"c",
"(c) hold-out",
hits,
len(questions),
THRESHOLD_HOLDOUT,
note=_HOLDOUT_NOTE,
misses=tuple(misses),
)
)
# (d) the same questions in plain Norwegian, fasit unchanged.
norwegian = _norwegian_questions(sets)
if norwegian is None:
rows.append(Row("d", "(d) Norwegian, asked directly", None, 0, THRESHOLD_NORWEGIAN_DIRECT))
else:
hits = 0
misses = []
for question_id, (asked, fasit) in norwegian.items():
excerpts = excerpts_of(ask(asked))
delivered_everywhere.extend(excerpts)
if question_hit(excerpts, fasit):
hits += 1
else:
misses.append(classify_miss(question_id, fasit, text))
rows.append(
Row(
"d",
"(d) Norwegian, asked directly",
hits,
len(norwegian),
THRESHOLD_NORWEGIAN_DIRECT,
misses=tuple(misses),
)
)
# (e) the same Norwegian questions, decomposed into English sub-questions.
if norwegian is None or sets.subquestions is None:
rows.append(
Row(
"e",
"(e) Norwegian, via sub-questions",
None,
0,
THRESHOLD_NORWEGIAN_SUBQUESTIONS,
note=_MERGE_NOTE,
)
)
else:
parts = sets.subquestions.get("delsporsmaal", {})
assert isinstance(parts, dict)
hits = 0
misses = []
for question_id, (_asked, fasit) in norwegian.items():
merged = excerpts_of(ask.many(parts.get(question_id, [])))
delivered_everywhere.extend(merged)
if question_hit(merged, fasit):
hits += 1
else:
misses.append(classify_miss(question_id, fasit, text))
rows.append(
Row(
"e",
"(e) Norwegian, via sub-questions",
hits,
len(norwegian),
THRESHOLD_NORWEGIAN_SUBQUESTIONS,
note=_MERGE_NOTE,
misses=tuple(misses),
)
)
# (f) the operator's own question, via the map-informed decomposition.
operator_note = ""
if sets.subquestions is None:
rows.append(Row("f", "(f) operator's question via OP_kart", None, 0, THRESHOLD_OPERATOR))
else:
operator = sets.subquestions.get("operator", {})
assert isinstance(operator, dict)
gold = operator.get("gold", [])
parts = sets.subquestions.get("delsporsmaal", {})
assert isinstance(gold, list) and isinstance(parts, dict)
direct = excerpts_of(ask(str(operator["question"])))
delivered_everywhere.extend(direct)
by_route: dict[str, int] = {}
for route in ("OP", "OP_kart"):
merged = excerpts_of(ask.many(parts.get(route, [])))
delivered_everywhere.extend(merged)
by_route[route] = sum(1 for place in gold if place_delivered(merged, place))
direct_places = sum(1 for place in gold if place_delivered(direct, place))
operator_note = (
f"asked directly: {direct_places} / {len(gold)}; via OP: {by_route['OP']} / {len(gold)}. "
f"declared hit_rule: {operator.get('hit_rule', '(none declared)')}"
)
rows.append(
Row(
"f",
"(f) operator's question via OP_kart",
by_route["OP_kart"],
len(gold),
THRESHOLD_OPERATOR,
note=operator_note,
)
)
# (g) the known negatives, and the same signal read over the English positives.
if sets.subquestions is None:
rows.append(
Row(
"g1",
"(g) negatives flagged",
None,
5,
THRESHOLD_NEGATIVE_FLAGGED,
note=_NEGATIVE_NOTE,
)
)
else:
negative = sets.subquestions.get("negative", {})
assert isinstance(negative, dict)
questions = negative.get("questions", [])
assert isinstance(questions, list)
flagged = 0
misses = []
for question in questions:
payload = ask(str(question["question"]))
delivered_everywhere.extend(excerpts_of(payload))
if uncovered_signal(payload):
flagged += 1
else:
misses.append(
Miss(
str(question["id"]),
"ikke flagget",
"the payload reads as an ordinary answer",
)
)
rows.append(
Row(
"g1",
"(g) negatives flagged",
flagged,
len(questions),
THRESHOLD_NEGATIVE_FLAGGED,
note=f"{_NEGATIVE_NOTE} declared pass_rule: {negative.get('pass_rule', '(none declared)')}",
misses=tuple(misses),
)
)
if not positives_complete:
rows.append(
Row(
"g2",
"(g) positives mis-flagged",
None,
0,
THRESHOLD_POSITIVE_MISFLAGGED,
at_most=True,
)
)
else:
misflagged = sum(1 for payload in english_positive_payloads if uncovered_signal(payload))
rows.append(
Row(
"g2",
"(g) positives mis-flagged",
misflagged,
len(english_positive_payloads),
THRESHOLD_POSITIVE_MISFLAGGED,
at_most=True,
note=_MISFLAG_NOTE,
)
)
# The largest delivered excerpt, over everything that ran: PM's finding
# that one concept of a real collection spends about a third of the budget
# by itself, so a single excerpt can crowd out the rest.
if not delivered_everywhere:
rows.append(
Row(
"h",
"largest delivered excerpt (chars)",
None,
0,
THRESHOLD_LARGEST_EXCERPT,
at_most=True,
)
)
notes: list[str] = []
else:
largest = max(delivered_everywhere, key=lambda excerpt: len(str(excerpt.get("text", ""))))
rows.append(
Row(
"h",
"largest delivered excerpt (chars)",
len(str(largest.get("text", ""))),
0,
THRESHOLD_LARGEST_EXCERPT,
at_most=True,
note=f"{largest.get('concept_id')} from {largest.get('source_file')}",
)
)
noise = sum(
1
for excerpt in delivered_everywhere
if NOISE_TITLE.match(str(excerpt.get("title", "")))
)
notes = [
f"delivered excerpts counted over every series that ran: {len(delivered_everywhere)}",
f"of those, titled `Tabell linje N` (PM's noise finding): {noise}",
]
return Report(collection=_collection_label(bundle_root), rows=rows, notes=tuple(notes))
_HOLDOUT_NOTE = (
"RUN AND REPORTED, NEVER TUNED AGAINST: a change that lifts (a) and not this "
"row learned the answer key. The bar is a floor, not a target."
)
_MERGE_NOTE = (
"the sub-questions are asked in ONE call and merged by the product "
"(`consume.build_multi_payload`), cut at the same k one question gets."
)
_NEGATIVE_NOTE = (
"the signal is `okf_retrieval_gate.marked`: nothing delivered, or the bundle "
"answers none of >= 2/3 of the question's own terms."
)
#: READ THIS ROW TOGETHER WITH (g). A low mis-flag count is cheap for a signal
#: that rarely fires at all, so this row can be green FOR THE SAME REASON (g)
#: is red. It is still worth its own row -- it is the only thing standing
#: between "say when you do not know" and a signal that says it about
#: everything -- but it is not evidence on its own.
_MISFLAG_NOTE = (
"green on its own means little while (g) is red: a signal that rarely fires "
"cannot often mis-fire. The pair is the measurement, not this row alone."
)
def _norwegian_questions(
sets: Sets,
) -> dict[str, tuple[str, Sequence[Mapping[str, str]]]] | None:
"""The Norwegian wording joined to the PHASE set's fasit, by id.
The fasit is unchanged by translation -- that is the whole point of the
series -- so it is read from the phase set and never duplicated into the
Norwegian file, where the two copies could drift.
"""
if sets.norwegian is None or sets.phase is None:
return None
asked = sets.norwegian.get("sporsmaal", {})
assert isinstance(asked, dict)
questions = sets.phase["questions"]
assert isinstance(questions, list)
fasit_by_id = {str(question["id"]): question["fasit"] for question in questions}
joined: dict[str, tuple[str, Sequence[Mapping[str, str]]]] = {}
for question_id, wording in asked.items():
fasit = fasit_by_id.get(str(question_id))
if fasit is None:
raise GateUsage(
f"norske-sporsmaal.json asks {question_id}, which fase-sporsmaal.json "
"does not carry a fasit for"
)
assert isinstance(fasit, list)
joined[str(question_id)] = (str(wording), fasit)
return joined
def _collection_label(bundle_root: Path) -> str:
"""The collection's own identity, never its path.
The table is pasted into STATE and a commit message; a scratch path in it
is noise that also makes two machines' output differ.
"""
return f"{consume.root_bundle_id_of(bundle_root)} @ {consume.bundle_ref(bundle_root)}"
# --- the command --------------------------------------------------------------
def parse_args(argv: Sequence[str] | None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
prog="okf-soek-gate",
description=(
"Measure what the asker RECEIVES from a collection, at the shipped "
"defaults, over the frozen question sets. Exit 0 only when every row holds."
),
)
parser.add_argument("--bundle", required=True, type=Path, help="the collection to measure")
parser.add_argument(
"--sets",
type=Path,
default=DEFAULT_SET_DIR,
help="the directory of frozen question sets (default: eval/soek/)",
)
return parser.parse_args(list(argv) if argv is not None else None)
def main(argv: Sequence[str] | None = None) -> int:
args = parse_args(argv)
try:
bundle_root = args.bundle
if not bundle_root.is_dir() or not (bundle_root / "index.md").is_file():
raise GateUsage(
f"no collection at {args.bundle}: build one first "
"(the command is in eval/soek/README.md). Refusing rather than "
"reporting 0 hits against nothing."
)
if not args.sets.is_dir():
raise GateUsage(f"no set directory at {args.sets}")
report = run(bundle_root, load_sets(args.sets))
except GateUsage as error:
print(f"okf-soek-gate: {error}", file=sys.stderr)
return 2
sys.stdout.write(report.render())
return report.exit_code()
if __name__ == "__main__":
raise SystemExit(main())