feat(consume): assemble a payload that measures its own emitted bytes
Two corrections the plan did not carry, both found by running the instrument: - A question matching nothing still returned the top eight by tie-break -- a confident guess wearing a denominator. The closed rule set gains no_lexical_match, so the known-negative returns a measured empty set with every considered concept named in withheld. - The stage-one document prior summed overlap across a document's units, so it measured document SIZE. Measured on K2 for the price question: the competition document sums to 6.0 over 79 concepts (0.076 each), the price document to 2.0 over 1. The prior is now a density; the price-form gold moves from outside the top eight to rank 1. Found with that answer visible, which the measurement document states beside the number. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
7eb87a04cb
commit
42ab9c1d1a
2 changed files with 417 additions and 25 deletions
|
|
@ -34,9 +34,11 @@ PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||||
sys.path.insert(0, str(PROJECT_ROOT / "tools"))
|
sys.path.insert(0, str(PROJECT_ROOT / "tools"))
|
||||||
|
|
||||||
import okf_consume # noqa: E402
|
import okf_consume # noqa: E402
|
||||||
|
import okf_contract_check # noqa: E402
|
||||||
|
|
||||||
from llm_ingestion_okf.materialize import parse_frontmatter # noqa: E402
|
from llm_ingestion_okf.materialize import parse_frontmatter # noqa: E402
|
||||||
|
|
||||||
|
TEMPLATE = PROJECT_ROOT / "skills" / "okf-consume-template" / "SKILL.md"
|
||||||
GOLDEN = PROJECT_ROOT / "examples" / "ingest-golden-segmented-okf-v0-2" / "expected-bundle"
|
GOLDEN = PROJECT_ROOT / "examples" / "ingest-golden-segmented-okf-v0-2" / "expected-bundle"
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -401,16 +403,16 @@ def test_concepts_tying_on_every_signal_come_back_in_concept_id_order() -> None:
|
||||||
# A question matching nothing makes every signal identical, so the ONLY
|
# A question matching nothing makes every signal identical, so the ONLY
|
||||||
# thing left deciding the order is the declared tie-break.
|
# thing left deciding the order is the declared tie-break.
|
||||||
ranked = okf_consume.concept_scores(concepts, "zzzz qqqq", {})
|
ranked = okf_consume.concept_scores(concepts, "zzzz qqqq", {})
|
||||||
ids = [concept.concept_id for concept, _ in ranked]
|
ids = [concept.concept_id for concept, _, _ in ranked]
|
||||||
assert ids == sorted(ids)
|
assert ids == sorted(ids)
|
||||||
|
|
||||||
|
|
||||||
def test_reversing_the_input_order_does_not_change_the_output_order() -> None:
|
def test_reversing_the_input_order_does_not_change_the_output_order() -> None:
|
||||||
concepts = _fixture_concepts()
|
concepts = _fixture_concepts()
|
||||||
forward = [c.concept_id for c, _ in okf_consume.concept_scores(concepts, "zzzz qqqq", {})]
|
forward = [c.concept_id for c, _, _ in okf_consume.concept_scores(concepts, "zzzz qqqq", {})]
|
||||||
backward = [
|
backward = [
|
||||||
c.concept_id
|
c.concept_id
|
||||||
for c, _ in okf_consume.concept_scores(list(reversed(concepts)), "zzzz qqqq", {})
|
for c, _, _ in okf_consume.concept_scores(list(reversed(concepts)), "zzzz qqqq", {})
|
||||||
]
|
]
|
||||||
assert forward == backward
|
assert forward == backward
|
||||||
|
|
||||||
|
|
@ -420,8 +422,8 @@ def test_a_concept_in_a_high_scoring_document_outranks_an_equally_lexical_one()
|
||||||
question = "Hvordan skal prisene fylles ut?"
|
question = "Hvordan skal prisene fylles ut?"
|
||||||
lifted = okf_consume.concept_scores(concepts, question, {"krav": 10.0, "dyp": 0.0})
|
lifted = okf_consume.concept_scores(concepts, question, {"krav": 10.0, "dyp": 0.0})
|
||||||
dropped = okf_consume.concept_scores(concepts, question, {"krav": 0.0, "dyp": 10.0})
|
dropped = okf_consume.concept_scores(concepts, question, {"krav": 0.0, "dyp": 10.0})
|
||||||
krav_first = [c.concept_id for c, _ in lifted].index("krav/prissammenstilling")
|
krav_first = [c.concept_id for c, _, _ in lifted].index("krav/prissammenstilling")
|
||||||
krav_later = [c.concept_id for c, _ in dropped].index("krav/prissammenstilling")
|
krav_later = [c.concept_id for c, _, _ in dropped].index("krav/prissammenstilling")
|
||||||
assert krav_first < krav_later
|
assert krav_first < krav_later
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -430,7 +432,7 @@ def test_the_ranked_order_is_identical_across_two_calls() -> None:
|
||||||
scores = okf_consume.document_scores(FIXTURE, "Hvordan skal prisene fylles ut?")
|
scores = okf_consume.document_scores(FIXTURE, "Hvordan skal prisene fylles ut?")
|
||||||
first = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores)
|
first = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores)
|
||||||
second = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores)
|
second = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores)
|
||||||
assert [c.concept_id for c, _ in first] == [c.concept_id for c, _ in second]
|
assert [c.concept_id for c, _, _ in first] == [c.concept_id for c, _, _ in second]
|
||||||
|
|
||||||
|
|
||||||
def test_the_price_concept_leads_on_the_price_question_in_the_fixture() -> None:
|
def test_the_price_concept_leads_on_the_price_question_in_the_fixture() -> None:
|
||||||
|
|
@ -472,15 +474,17 @@ def test_a_verdict_concept_is_withheld_by_rule_and_reaches_no_excerpt() -> None:
|
||||||
|
|
||||||
|
|
||||||
def test_the_verdict_exclusion_is_a_type_check_and_never_a_path_filter() -> None:
|
def test_the_verdict_exclusion_is_a_type_check_and_never_a_path_filter() -> None:
|
||||||
# A `type: reference` file NAMED `verdict-lookalike` is delivered; a
|
# One question reaching BOTH: a `type: reference` file NAMED
|
||||||
# `type: verdict` file under an ordinary name at depth 3 is withheld. A path
|
# `verdict-lookalike` is delivered, and a `type: verdict` file under an
|
||||||
# filter gets both backwards.
|
# ordinary name at depth 3 is withheld. A path filter gets both backwards,
|
||||||
delivered, withheld, _ = _cut_fixture()
|
# and neither half of this is measured unless both concepts match.
|
||||||
|
delivered, withheld, _ = _cut_fixture(question="Hva sier notatet om stifilter og typesjekk?")
|
||||||
delivered_ids = {excerpt["concept_id"] for excerpt in delivered}
|
delivered_ids = {excerpt["concept_id"] for excerpt in delivered}
|
||||||
rules = dict(withheld)
|
rules = dict(withheld)
|
||||||
assert "krav/verdict-lookalike" in delivered_ids
|
assert "krav/verdict-lookalike" in delivered_ids
|
||||||
assert "dyp/nivaa/alminnelig-notat" not in delivered_ids
|
assert "dyp/nivaa/alminnelig-notat" not in delivered_ids
|
||||||
assert rules["dyp/nivaa/alminnelig-notat"] == "verdict_layer_excluded"
|
assert rules["dyp/nivaa/alminnelig-notat"] == "verdict_layer_excluded"
|
||||||
|
assert rules["dyp/nivaa/alminnelig-notat"] != "no_lexical_match"
|
||||||
|
|
||||||
|
|
||||||
def test_a_capital_l_log_type_does_not_crash_the_reader() -> None:
|
def test_a_capital_l_log_type_does_not_crash_the_reader() -> None:
|
||||||
|
|
@ -495,7 +499,9 @@ def test_a_concept_whose_verified_cannot_be_read_is_withheld_by_name() -> None:
|
||||||
# SS 6.2 requires a tier on every excerpt and SS 6.4 forbids reading absence
|
# SS 6.2 requires a tier on every excerpt and SS 6.4 forbids reading absence
|
||||||
# as negation. Emitting `unverified` for an unreadable value asserts a fact
|
# as negation. Emitting `unverified` for an unreadable value asserts a fact
|
||||||
# nobody measured.
|
# nobody measured.
|
||||||
_, withheld, _ = _cut_fixture()
|
# The question must REACH the concept: a relevance drop fires first, and a
|
||||||
|
# `no_lexical_match` here would prove nothing about tiering.
|
||||||
|
_, withheld, _ = _cut_fixture(question="Hva staar i blokkform?")
|
||||||
assert dict(withheld)["dyp/nivaa/blokkform-verifisert"] == "verified_unreadable"
|
assert dict(withheld)["dyp/nivaa/blokkform-verifisert"] == "verified_unreadable"
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -523,8 +529,13 @@ def test_a_concept_larger_than_the_limit_is_excluded_by_name_before_the_dp() ->
|
||||||
|
|
||||||
|
|
||||||
def test_concepts_ranked_beyond_k_are_withheld_as_below_k() -> None:
|
def test_concepts_ranked_beyond_k_are_withheld_as_below_k() -> None:
|
||||||
_, withheld, _ = _cut_fixture(k=1)
|
# A question matching TWO concepts, so that capping at one leaves a real
|
||||||
assert "below_k" in {rule for _, rule in withheld}
|
# `below_k` drop rather than an empty set.
|
||||||
|
matching = "kontroll av prisene"
|
||||||
|
_, wide, _ = _cut_fixture(question=matching, k=8)
|
||||||
|
assert "below_k" not in {rule for _, rule in wide}
|
||||||
|
_, narrow, _ = _cut_fixture(question=matching, k=1)
|
||||||
|
assert "below_k" in {rule for _, rule in narrow}
|
||||||
|
|
||||||
|
|
||||||
def test_the_exact_knapsack_beats_greedy_by_density() -> None:
|
def test_the_exact_knapsack_beats_greedy_by_density() -> None:
|
||||||
|
|
@ -549,3 +560,215 @@ def test_excerpts_come_back_in_rank_order_and_carry_that_rank() -> None:
|
||||||
delivered, _, _ = _cut_fixture()
|
delivered, _, _ = _cut_fixture()
|
||||||
assert [excerpt["rank"] for excerpt in delivered] == list(range(1, len(delivered) + 1))
|
assert [excerpt["rank"] for excerpt in delivered] == list(range(1, len(delivered) + 1))
|
||||||
assert delivered[0]["concept_id"] == "krav/prissammenstilling"
|
assert delivered[0]["concept_id"] == "krav/prissammenstilling"
|
||||||
|
|
||||||
|
|
||||||
|
# --- Step 8: the payload ------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def _payload(
|
||||||
|
root: Path = FIXTURE, question: str = "Hvordan skal prisene fylles ut?", **kwargs: object
|
||||||
|
) -> dict[str, object]:
|
||||||
|
return okf_consume.build_payload(root, question=question, **kwargs) # type: ignore[arg-type]
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_payload_passes_the_checker_against_the_template_with_zero_findings() -> None:
|
||||||
|
report = okf_contract_check.check(TEMPLATE.read_text(encoding="utf-8"), _payload())
|
||||||
|
assert report.findings == ()
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_payload_carries_every_section_eight_member() -> None:
|
||||||
|
payload = _payload()
|
||||||
|
assert payload["contract"] == "okf-consumption/1"
|
||||||
|
assert set(payload) >= {
|
||||||
|
"contract",
|
||||||
|
"bundle",
|
||||||
|
"budget",
|
||||||
|
"denominators",
|
||||||
|
"excerpts",
|
||||||
|
"withheld",
|
||||||
|
}
|
||||||
|
bundle = payload["bundle"]
|
||||||
|
assert isinstance(bundle, dict)
|
||||||
|
assert bundle["bundle_id"] == "consume-fixture"
|
||||||
|
assert str(bundle["ref"]).startswith("sha256-tree:")
|
||||||
|
|
||||||
|
|
||||||
|
def test_spent_is_the_cost_of_the_delivered_set_and_not_of_the_whole_payload() -> None:
|
||||||
|
# SS 7.2 verbatim: "what the DELIVERED SET spent by that same instrument".
|
||||||
|
# Measured on K2 at k=8, the whole-payload reading puts a 628-entry
|
||||||
|
# `withheld` list (81 565 B) plus one gold excerpt (101 576 B) against a
|
||||||
|
# 120 000 B limit -- so a CORRECT implementation would exit 1 and fail its
|
||||||
|
# own SC1 and SC6.
|
||||||
|
payload = _payload()
|
||||||
|
budget = payload["budget"]
|
||||||
|
excerpts = payload["excerpts"]
|
||||||
|
assert isinstance(budget, dict) and isinstance(excerpts, list)
|
||||||
|
assert budget["spent"] == sum(okf_consume.excerpt_weight(e) for e in excerpts)
|
||||||
|
|
||||||
|
|
||||||
|
def test_spent_moves_when_an_excerpt_moves_and_holds_when_withheld_grows() -> None:
|
||||||
|
# The property that distinguishes SS 7.2's reading from the whole-payload
|
||||||
|
# one, asserted rather than described.
|
||||||
|
matching = "kontroll av prisene"
|
||||||
|
wide = _payload(question=matching, k=8)
|
||||||
|
narrow = _payload(question=matching, k=1)
|
||||||
|
wide_budget, narrow_budget = wide["budget"], narrow["budget"]
|
||||||
|
wide_counts, narrow_counts = wide["denominators"], narrow["denominators"]
|
||||||
|
assert isinstance(wide_budget, dict) and isinstance(narrow_budget, dict)
|
||||||
|
assert isinstance(wide_counts, dict) and isinstance(narrow_counts, dict)
|
||||||
|
assert narrow_counts["withheld"] > wide_counts["withheld"]
|
||||||
|
assert narrow_budget["spent"] < wide_budget["spent"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_counts_and_the_lists_are_two_statements_of_one_fact() -> None:
|
||||||
|
payload = _payload()
|
||||||
|
counts, excerpts, withheld = payload["denominators"], payload["excerpts"], payload["withheld"]
|
||||||
|
assert isinstance(counts, dict) and isinstance(excerpts, list) and isinstance(withheld, list)
|
||||||
|
assert counts["delivered"] == len(excerpts)
|
||||||
|
assert counts["withheld"] == len(withheld)
|
||||||
|
assert counts["considered"] == counts["delivered"] + counts["withheld"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_excerpt_digest_recomputes_from_the_named_concepts_bytes() -> None:
|
||||||
|
payload = _payload()
|
||||||
|
excerpts = payload["excerpts"]
|
||||||
|
assert isinstance(excerpts, list) and excerpts
|
||||||
|
for excerpt in excerpts:
|
||||||
|
on_disk = FIXTURE / f"{excerpt['concept_id']}.md"
|
||||||
|
assert excerpt["sha256"] == hashlib.sha256(on_disk.read_bytes()).hexdigest()
|
||||||
|
assert (
|
||||||
|
excerpt["text_sha256"]
|
||||||
|
== hashlib.sha256(str(excerpt["text"]).encode("utf-8")).hexdigest()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_concept_id_keeps_the_slash_import_slug_would_have_removed() -> None:
|
||||||
|
payload = _payload()
|
||||||
|
excerpts = payload["excerpts"]
|
||||||
|
assert isinstance(excerpts, list)
|
||||||
|
assert any("/" in str(excerpt["concept_id"]) for excerpt in excerpts)
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_budget_refuses_rather_than_narrowing_when_the_cut_cannot_fit() -> None:
|
||||||
|
# SS 7.3: exceeding the gate is a finding requiring a decision, never
|
||||||
|
# something to retry narrower.
|
||||||
|
with pytest.raises(okf_consume.ConsumeError) as raised:
|
||||||
|
_payload(limit=100)
|
||||||
|
assert raised.value.code == "budget_admits_nothing"
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_serialised_payload_is_lf_only_and_ends_in_exactly_one_newline() -> None:
|
||||||
|
text = okf_consume.serialise(_payload())
|
||||||
|
assert "\r" not in text
|
||||||
|
assert text.endswith("}\n")
|
||||||
|
assert not text.endswith("}\n\n")
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_serialised_payload_does_not_escape_norwegian_letters() -> None:
|
||||||
|
# `ensure_ascii=True` inflates this corpus by 7.1 %, which is more than the
|
||||||
|
# headroom the gate leaves.
|
||||||
|
text = okf_consume.serialise(_payload(question="Hvor ofte er den årlige kontrollen?"))
|
||||||
|
assert "\\u00e5" not in text
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_question_with_no_answer_returns_a_measured_empty_set_not_a_guess() -> None:
|
||||||
|
# The order's known-negative control, and the reason the `no_lexical_match`
|
||||||
|
# rule exists: a ranker that always returns its top eight scores well on
|
||||||
|
# every positive question and is useless. The emptiness must be POSITIVE --
|
||||||
|
# every considered concept named in `withheld` under a rule, so the identity
|
||||||
|
# still closes and the skill can say "measured, nothing cleared the bar"
|
||||||
|
# rather than "nothing was found".
|
||||||
|
payload = _payload(question="Hva er reglene for sveising av titan i vakuum?")
|
||||||
|
counts, excerpts, withheld = payload["denominators"], payload["excerpts"], payload["withheld"]
|
||||||
|
assert isinstance(counts, dict) and isinstance(excerpts, list) and isinstance(withheld, list)
|
||||||
|
assert excerpts == []
|
||||||
|
assert counts["delivered"] == 0
|
||||||
|
assert (
|
||||||
|
counts["withheld"] == counts["considered"] == len(okf_consume.enumerate_concepts(FIXTURE))
|
||||||
|
)
|
||||||
|
assert {entry["rule"] for entry in withheld} == {"no_lexical_match", "verdict_layer_excluded"}
|
||||||
|
# And the control: the SAME payload builder returns a non-empty set for a
|
||||||
|
# question this bundle does answer, so the zero is a measurement.
|
||||||
|
answered = _payload()
|
||||||
|
answered_excerpts = answered["excerpts"]
|
||||||
|
assert isinstance(answered_excerpts, list) and answered_excerpts
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_empty_payload_still_passes_the_checker() -> None:
|
||||||
|
payload = _payload(question="Hva er reglene for sveising av titan i vakuum?")
|
||||||
|
assert okf_contract_check.check(TEMPLATE.read_text(encoding="utf-8"), payload).findings == ()
|
||||||
|
|
||||||
|
|
||||||
|
# --- Corpus-conditional arms --------------------------------------------------
|
||||||
|
|
||||||
|
K2_BUNDLE = Path.home() / "corpora" / "okf-telling-20260829" / "K2-bundle-20260903"
|
||||||
|
K2_CONCEPTS = 629
|
||||||
|
K2_PROPOSED = 618
|
||||||
|
K2_KEYLESS = 11
|
||||||
|
|
||||||
|
requires_k2 = pytest.mark.skipif(
|
||||||
|
not K2_BUNDLE.is_dir(),
|
||||||
|
reason=(
|
||||||
|
f"the K2 corpus is not present at {K2_BUNDLE}. NOT MEASURED, not zero: "
|
||||||
|
f"this arm covers a denominator of {K2_CONCEPTS} concepts, of which "
|
||||||
|
f"{K2_PROPOSED} carry `adjudication: proposed` and {K2_KEYLESS} carry no "
|
||||||
|
"`adjudication` key at all. A skip here is an unmeasured denominator, "
|
||||||
|
"never a pass."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@requires_k2
|
||||||
|
def test_the_eleven_keyless_k2_concepts_come_back_unknown_over_a_stated_denominator() -> None:
|
||||||
|
# SS 6.1's third state, on real data rather than on a fixture. The 11 are
|
||||||
|
# asserted as ONE named set: measured, the concepts carrying no
|
||||||
|
# `adjudication` are EXACTLY those carrying no `bundle_id`, so three
|
||||||
|
# independent counts would share one blind spot.
|
||||||
|
root_bundle_id = parse_frontmatter(K2_BUNDLE / "index.md")["bundle_id"]
|
||||||
|
concepts = [
|
||||||
|
okf_consume.read_concept(
|
||||||
|
K2_BUNDLE / f"{concept_id}.md",
|
||||||
|
bundle_root=K2_BUNDLE,
|
||||||
|
root_bundle_id=root_bundle_id,
|
||||||
|
)
|
||||||
|
for concept_id in okf_consume.enumerate_concepts(K2_BUNDLE)
|
||||||
|
]
|
||||||
|
assert len(concepts) == K2_CONCEPTS
|
||||||
|
unknown = {c.concept_id for c in concepts if c.adjudication == "unknown"}
|
||||||
|
inherited = {c.concept_id for c in concepts if c.bundle_id_inherited}
|
||||||
|
proposed = [c for c in concepts if c.adjudication == "proposed"]
|
||||||
|
assert len(proposed) == K2_PROPOSED
|
||||||
|
assert len(unknown) == K2_KEYLESS
|
||||||
|
assert unknown == inherited, "the two sets diverged; the fallback is no longer one fact"
|
||||||
|
assert all(c.bundle_id == root_bundle_id for c in concepts if c.bundle_id_inherited)
|
||||||
|
# `adjudicated` has denominator ZERO on this corpus. Stated, not implied.
|
||||||
|
assert [c for c in concepts if c.adjudication == "adjudicated"] == []
|
||||||
|
|
||||||
|
|
||||||
|
@requires_k2
|
||||||
|
def test_spent_is_the_delivered_set_where_the_whole_payload_reading_would_refuse() -> None:
|
||||||
|
# The regression guard, with figures RE-MEASURED here rather than carried
|
||||||
|
# from the plan: the plan predicted 101 576 B for this excerpt and 188 758 B
|
||||||
|
# for the payload, both taken before per-line trailing-whitespace stripping
|
||||||
|
# landed. What this build actually produces is recorded instead.
|
||||||
|
payload = okf_consume.build_payload(K2_BUNDLE, question="Hvordan skal prisene fylles ut?")
|
||||||
|
budget, excerpts = payload["budget"], payload["excerpts"]
|
||||||
|
assert isinstance(budget, dict) and isinstance(excerpts, list)
|
||||||
|
whole_payload = len(okf_consume.serialise(payload).encode("utf-8"))
|
||||||
|
assert whole_payload > int(budget["limit"]), (
|
||||||
|
"the guard measures nothing: the whole payload already fits, so the two "
|
||||||
|
"readings of SS 7.2 cannot be told apart on this case"
|
||||||
|
)
|
||||||
|
assert int(budget["spent"]) <= int(budget["limit"])
|
||||||
|
|
||||||
|
|
||||||
|
@requires_k2
|
||||||
|
def test_the_price_form_gold_is_delivered_for_the_price_question() -> None:
|
||||||
|
# SC6. The one question whose gold is confirmed by a signal from outside
|
||||||
|
# this repository: a live model reached that directory unprompted in three
|
||||||
|
# navigation steps on 2026-09-06.
|
||||||
|
payload = okf_consume.build_payload(K2_BUNDLE, question="Hvordan skal prisene fylles ut?")
|
||||||
|
excerpts = payload["excerpts"]
|
||||||
|
assert isinstance(excerpts, list)
|
||||||
|
ids = [str(excerpt["concept_id"]) for excerpt in excerpts]
|
||||||
|
assert "del-ii-bilag-7-prisskjema/prissammenstilling-sheet-1" in ids
|
||||||
|
|
|
||||||
|
|
@ -550,7 +550,7 @@ def _overlap(question_tokens: Sequence[str], candidate: str) -> int:
|
||||||
def document_scores(
|
def document_scores(
|
||||||
bundle_root: Path, question: str, *, profile: BundleProfile = DEFAULT_PROFILE
|
bundle_root: Path, question: str, *, profile: BundleProfile = DEFAULT_PROFILE
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
"""One score per top-level document, from the indexes and the path alone.
|
"""One score per top-level document, from the indexes and the paths alone.
|
||||||
|
|
||||||
A "document" is a top-level entry: a directory, or -- per the corpus's own
|
A "document" is a top-level entry: a directory, or -- per the corpus's own
|
||||||
shape -- a concept sitting at the root, which has no directory to inherit
|
shape -- a concept sitting at the root, which has no directory to inherit
|
||||||
|
|
@ -559,27 +559,45 @@ def document_scores(
|
||||||
nor `bundle_id`; scoring them as members of some parent would put one bug in
|
nor `bundle_id`; scoring them as members of some parent would put one bug in
|
||||||
three places.
|
three places.
|
||||||
|
|
||||||
|
**The score is a DENSITY, not a sum, and that is a correction rather than a
|
||||||
|
preference.** A sum over a document's units grows with the number of units,
|
||||||
|
so a large document outscores a small one on size alone. Measured on K2 for
|
||||||
|
the price question: the competition document sums to 6.0 over 79 concepts
|
||||||
|
(0.076 each) and the price document to 2.0 over 1 (2.0), so the sum ranks
|
||||||
|
the larger document three times higher while the density ranks the smaller
|
||||||
|
one twenty-six times higher. A prior that grows with size is measuring size.
|
||||||
|
|
||||||
|
**Stated because it bears on how the hit@k number should be read:** this
|
||||||
|
defect was found by running the one question whose gold was confirmed
|
||||||
|
independently, and fixing it therefore happened with that answer visible.
|
||||||
|
The fix is justified by the scoring function's own arithmetic rather than by
|
||||||
|
the answer -- but the ranker is not blind to that one case, and the
|
||||||
|
measurement document says so beside the number.
|
||||||
|
|
||||||
Reads the INDEX TREE only. No directory is enumerated here or anywhere else
|
Reads the INDEX TREE only. No directory is enumerated here or anywhere else
|
||||||
in this command (SS 9.2).
|
in this command (SS 9.2).
|
||||||
"""
|
"""
|
||||||
question_tokens = normalise(question)
|
question_tokens = normalise(question)
|
||||||
indexes, concepts = _walk_index_tree(bundle_root, profile=profile)
|
indexes, concepts = _walk_index_tree(bundle_root, profile=profile)
|
||||||
scores: dict[str, float] = {}
|
totals: dict[str, float] = {}
|
||||||
|
units: dict[str, int] = {}
|
||||||
|
|
||||||
|
def record(document: str, overlap: int) -> None:
|
||||||
|
totals[document] = totals.get(document, 0.0) + float(overlap)
|
||||||
|
units[document] = units.get(document, 0) + 1
|
||||||
|
|
||||||
for concept_id in concepts:
|
for concept_id in concepts:
|
||||||
document = concept_id.split("/", 1)[0]
|
record(concept_id.split("/", 1)[0], _overlap(question_tokens, concept_id.replace("/", " ")))
|
||||||
scores.setdefault(document, 0.0)
|
|
||||||
scores[document] += float(_overlap(question_tokens, concept_id.replace("/", " ")))
|
|
||||||
for relative in indexes:
|
for relative in indexes:
|
||||||
document = relative.split("/", 1)[0]
|
document = relative.split("/", 1)[0]
|
||||||
if document == profile.index.name:
|
if document == profile.index.name:
|
||||||
continue
|
continue
|
||||||
scores.setdefault(document, 0.0)
|
|
||||||
for line in (bundle_root / relative).read_text(encoding="utf-8").splitlines():
|
for line in (bundle_root / relative).read_text(encoding="utf-8").splitlines():
|
||||||
entry = profile.index.parse_entry(line)
|
entry = profile.index.parse_entry(line)
|
||||||
if entry is None:
|
if entry is None:
|
||||||
continue
|
continue
|
||||||
scores[document] += float(_overlap(question_tokens, entry.label))
|
record(document, _overlap(question_tokens, entry.label))
|
||||||
return scores
|
return {document: totals[document] / units[document] for document in totals}
|
||||||
|
|
||||||
|
|
||||||
# --- Stage two: which concepts inside those documents -------------------------
|
# --- Stage two: which concepts inside those documents -------------------------
|
||||||
|
|
@ -600,7 +618,7 @@ def concept_scores(
|
||||||
concepts: Sequence[Concept],
|
concepts: Sequence[Concept],
|
||||||
question: str,
|
question: str,
|
||||||
document_score: Mapping[str, float],
|
document_score: Mapping[str, float],
|
||||||
) -> list[tuple[Concept, float]]:
|
) -> list[tuple[Concept, float, int]]:
|
||||||
"""Every concept, ordered best first, fused from three signals by RRF.
|
"""Every concept, ordered best first, fused from three signals by RRF.
|
||||||
|
|
||||||
The signals: (1) the question against the concept's title and the segments
|
The signals: (1) the question against the concept's title and the segments
|
||||||
|
|
@ -614,6 +632,12 @@ def concept_scores(
|
||||||
|
|
||||||
**No float reaches the payload.** These scores order the cut; only ranks and
|
**No float reaches the payload.** These scores order the cut; only ranks and
|
||||||
whole byte counts are emitted.
|
whole byte counts are emitted.
|
||||||
|
|
||||||
|
The third element of each tuple is the concept's OWN lexical overlap --
|
||||||
|
signals 1 and 2 only, with the document prior excluded. The cut needs it
|
||||||
|
separately: a concept that answers nothing in the question, sitting in a
|
||||||
|
document that does, is a GUESS, and a guess is the one thing a declared cut
|
||||||
|
must not deliver.
|
||||||
"""
|
"""
|
||||||
question_tokens = normalise(question)
|
question_tokens = normalise(question)
|
||||||
signals: list[dict[str, float]] = [
|
signals: list[dict[str, float]] = [
|
||||||
|
|
@ -639,9 +663,15 @@ def concept_scores(
|
||||||
order = sorted(signal, key=lambda key: (-signal[key], key))
|
order = sorted(signal, key=lambda key: (-signal[key], key))
|
||||||
for position, concept_id in enumerate(order, start=1):
|
for position, concept_id in enumerate(order, start=1):
|
||||||
fused[concept_id] += 1.0 / (RRF_K + position)
|
fused[concept_id] += 1.0 / (RRF_K + position)
|
||||||
|
lexical = {
|
||||||
|
concept.concept_id: int(signals[0][concept.concept_id] + signals[1][concept.concept_id])
|
||||||
|
for concept in concepts
|
||||||
|
}
|
||||||
by_id = {concept.concept_id: concept for concept in concepts}
|
by_id = {concept.concept_id: concept for concept in concepts}
|
||||||
ranked_ids = sorted(fused, key=lambda key: (-fused[key], key))
|
ranked_ids = sorted(fused, key=lambda key: (-fused[key], key))
|
||||||
return [(by_id[concept_id], fused[concept_id]) for concept_id in ranked_ids]
|
return [
|
||||||
|
(by_id[concept_id], fused[concept_id], lexical[concept_id]) for concept_id in ranked_ids
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
# --- The cut (SS 5, SS 7, SS 9.1) ---------------------------------------------
|
# --- The cut (SS 5, SS 7, SS 9.1) ---------------------------------------------
|
||||||
|
|
@ -652,6 +682,7 @@ def concept_scores(
|
||||||
WITHHOLDING_RULES = (
|
WITHHOLDING_RULES = (
|
||||||
"verdict_layer_excluded",
|
"verdict_layer_excluded",
|
||||||
"verified_unreadable",
|
"verified_unreadable",
|
||||||
|
"no_lexical_match",
|
||||||
"over_budget_alone",
|
"over_budget_alone",
|
||||||
"below_k",
|
"below_k",
|
||||||
"over_budget_after_knapsack",
|
"over_budget_after_knapsack",
|
||||||
|
|
@ -733,7 +764,7 @@ def knapsack(items: Sequence[tuple[float, int]], *, capacity: int) -> tuple[int,
|
||||||
|
|
||||||
|
|
||||||
def cut(
|
def cut(
|
||||||
ranked: Sequence[tuple[Concept, float]], *, k: int, limit: int
|
ranked: Sequence[tuple[Concept, float, int]], *, k: int, limit: int
|
||||||
) -> tuple[tuple[dict[str, object], ...], tuple[tuple[str, str], ...]]:
|
) -> tuple[tuple[dict[str, object], ...], tuple[tuple[str, str], ...]]:
|
||||||
"""The ranked concepts split into delivered excerpts and named drops.
|
"""The ranked concepts split into delivered excerpts and named drops.
|
||||||
|
|
||||||
|
|
@ -743,16 +774,25 @@ def cut(
|
||||||
|
|
||||||
Exclusions run before the pack, each naming its rule, because "it did not
|
Exclusions run before the pack, each naming its rule, because "it did not
|
||||||
fit" and "it could never be delivered" are different facts about the cut.
|
fit" and "it could never be delivered" are different facts about the cut.
|
||||||
|
|
||||||
|
**A concept answering nothing in the question is withheld, never ranked into
|
||||||
|
the top k as filler.** Without that rule a question with no answer in the
|
||||||
|
bundle still returns eight excerpts -- a confident guess wearing a
|
||||||
|
denominator -- and hit@k over such a ranker measures the corpus's size
|
||||||
|
rather than the ranker.
|
||||||
"""
|
"""
|
||||||
withheld: list[tuple[str, str]] = []
|
withheld: list[tuple[str, str]] = []
|
||||||
candidates: list[tuple[Concept, float, dict[str, object], int]] = []
|
candidates: list[tuple[Concept, float, dict[str, object], int]] = []
|
||||||
for concept, score in ranked:
|
for concept, score, lexical in ranked:
|
||||||
# SS 9.1: a TYPE check at every level, never a path filter. Case-folded
|
# SS 9.1: a TYPE check at every level, never a path filter. Case-folded
|
||||||
# against the profile's own constant, because `type: Log` (capital L)
|
# against the profile's own constant, because `type: Log` (capital L)
|
||||||
# really occurs in the corpus.
|
# really occurs in the corpus.
|
||||||
if concept.okf_type.casefold() == RESERVED_OKF_TYPE:
|
if concept.okf_type.casefold() == RESERVED_OKF_TYPE:
|
||||||
withheld.append((concept.concept_id, "verdict_layer_excluded"))
|
withheld.append((concept.concept_id, "verdict_layer_excluded"))
|
||||||
continue
|
continue
|
||||||
|
if lexical == 0:
|
||||||
|
withheld.append((concept.concept_id, "no_lexical_match"))
|
||||||
|
continue
|
||||||
excerpt = excerpt_for(concept)
|
excerpt = excerpt_for(concept)
|
||||||
if excerpt is None:
|
if excerpt is None:
|
||||||
withheld.append((concept.concept_id, "verified_unreadable"))
|
withheld.append((concept.concept_id, "verified_unreadable"))
|
||||||
|
|
@ -788,3 +828,132 @@ def cut(
|
||||||
withheld.append((concept.concept_id, "over_budget_after_knapsack"))
|
withheld.append((concept.concept_id, "over_budget_after_knapsack"))
|
||||||
withheld.sort()
|
withheld.sort()
|
||||||
return tuple(delivered), tuple(withheld)
|
return tuple(delivered), tuple(withheld)
|
||||||
|
|
||||||
|
|
||||||
|
# --- The payload (SS 8) -------------------------------------------------------
|
||||||
|
|
||||||
|
#: SS 8.2: present so a reader can tell which revision it is holding.
|
||||||
|
CONTRACT_REVISION = "okf-consumption/1"
|
||||||
|
|
||||||
|
#: `--k` caps the DELIVERED set. The budget is the gate; this is a second,
|
||||||
|
#: cheaper bound so a question matching half the corpus does not run a
|
||||||
|
#: 300-item DP to discover the same answer.
|
||||||
|
DEFAULT_K = 8
|
||||||
|
|
||||||
|
|
||||||
|
def build_payload(
|
||||||
|
bundle_root: Path,
|
||||||
|
*,
|
||||||
|
question: str,
|
||||||
|
k: int = DEFAULT_K,
|
||||||
|
limit: int = DEFAULT_LIMIT,
|
||||||
|
profile: BundleProfile = DEFAULT_PROFILE,
|
||||||
|
) -> dict[str, object]:
|
||||||
|
"""One bundle plus one question, cut to one contract-conformant payload.
|
||||||
|
|
||||||
|
Pure with respect to the clock and the network: the same
|
||||||
|
`(bundle_root, question, k, limit)` at the same bytes returns the same
|
||||||
|
object, every time.
|
||||||
|
"""
|
||||||
|
case, expected, measured = known_positive()
|
||||||
|
if expected != measured:
|
||||||
|
# SS 7.4: the instrument reports NONE of its own numbers until it has
|
||||||
|
# reproduced a known figure. Refusing here rather than emitting a
|
||||||
|
# payload with a failing known-positive is the difference between an
|
||||||
|
# instrument that has been shown to count and one that merely says so.
|
||||||
|
raise ConsumeError(
|
||||||
|
f"the budget instrument's known-positive ({case}) expected {expected} "
|
||||||
|
f"and measured {measured}; refusing to report any figure until the "
|
||||||
|
"two agree",
|
||||||
|
code="instrument_unvalidated",
|
||||||
|
)
|
||||||
|
root_index = bundle_root / profile.index.name
|
||||||
|
if not root_index.is_file():
|
||||||
|
raise ConsumeError(
|
||||||
|
f"{bundle_root} carries no {profile.index.name}, so there is no index "
|
||||||
|
"tree to walk and no way to read the bundle without enumerating a "
|
||||||
|
"directory, which SS 9.2 forbids",
|
||||||
|
code="bundle_unreadable",
|
||||||
|
)
|
||||||
|
root_bundle_id = parse_frontmatter(root_index).get("bundle_id", "")
|
||||||
|
if not root_bundle_id:
|
||||||
|
raise ConsumeError(
|
||||||
|
f"{root_index} declares no `bundle_id`; identity across bundles is "
|
||||||
|
"the (bundle_id, concept_id) tuple (SS 3.1) and half of it is missing",
|
||||||
|
code="bundle_id_missing",
|
||||||
|
)
|
||||||
|
concept_ids = enumerate_concepts(bundle_root, profile=profile)
|
||||||
|
concepts = [
|
||||||
|
read_concept(
|
||||||
|
bundle_root / f"{concept_id}{profile.paths.concept_suffix}",
|
||||||
|
bundle_root=bundle_root,
|
||||||
|
root_bundle_id=root_bundle_id,
|
||||||
|
)
|
||||||
|
for concept_id in concept_ids
|
||||||
|
]
|
||||||
|
ranked = concept_scores(
|
||||||
|
concepts, question, document_scores(bundle_root, question, profile=profile)
|
||||||
|
)
|
||||||
|
matched = sum(1 for _, _, lexical in ranked if lexical > 0)
|
||||||
|
delivered, withheld = cut(ranked, k=k, limit=limit)
|
||||||
|
spent = sum(excerpt_weight(excerpt) for excerpt in delivered)
|
||||||
|
if matched and not delivered:
|
||||||
|
# SS 7.3: a finding requiring a decision, never something to retry
|
||||||
|
# narrower. Distinguished from the honest empty result BY THE
|
||||||
|
# DENOMINATOR: concepts answered this question and the budget admitted
|
||||||
|
# none of them, which is a statement about the limit, not about the
|
||||||
|
# bundle.
|
||||||
|
raise ConsumeError(
|
||||||
|
f"{matched} concept(s) answered this question and the {limit}-byte "
|
||||||
|
f"budget admitted none of them; the cut strategy is wrong for this "
|
||||||
|
"bundle at this limit -- refusing rather than emitting an empty "
|
||||||
|
"payload that would read as 'nothing was found'",
|
||||||
|
code="budget_admits_nothing",
|
||||||
|
)
|
||||||
|
if spent > limit:
|
||||||
|
# Structurally unreachable while the knapsack bucket arithmetic is
|
||||||
|
# one-sided, and kept because SS 7.3 is a MUST about the emitted
|
||||||
|
# payload rather than about the algorithm that produced it.
|
||||||
|
raise ConsumeError(f"spent ({spent}) exceeds limit ({limit})", code="budget_exceeded")
|
||||||
|
return {
|
||||||
|
"contract": CONTRACT_REVISION,
|
||||||
|
"bundle": {
|
||||||
|
"bundle_id": root_bundle_id,
|
||||||
|
"ref": bundle_ref(bundle_root, profile=profile),
|
||||||
|
},
|
||||||
|
"budget": {
|
||||||
|
"unit": BUDGET_UNIT,
|
||||||
|
"instrument": BUDGET_INSTRUMENT,
|
||||||
|
"limit": limit,
|
||||||
|
"spent": spent,
|
||||||
|
"known_positive": {
|
||||||
|
"case": case,
|
||||||
|
"expected": expected,
|
||||||
|
"measured": measured,
|
||||||
|
# The second, independent route (`wc -c` on the same file), so
|
||||||
|
# `expected == measured` is not the only thing standing between
|
||||||
|
# a broken instrument and a green gate.
|
||||||
|
"raw_bytes": len(_KNOWN_POSITIVE_PATH.read_bytes()),
|
||||||
|
"encoding_delta": KNOWN_POSITIVE_ENCODING_DELTA,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"denominators": {
|
||||||
|
"considered": len(concepts),
|
||||||
|
"withheld": len(withheld),
|
||||||
|
"delivered": len(delivered),
|
||||||
|
},
|
||||||
|
"question": question,
|
||||||
|
"excerpts": list(delivered),
|
||||||
|
"withheld": [{"concept_id": concept_id, "rule": rule} for concept_id, rule in withheld],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def serialise(payload: Mapping[str, object]) -> str:
|
||||||
|
"""The payload as the bytes that are actually emitted.
|
||||||
|
|
||||||
|
`ensure_ascii=False` is load-bearing rather than cosmetic: the default
|
||||||
|
inflates this corpus by 7.1 % (1 950 745 -> 2 089 391 B), so a gate
|
||||||
|
measuring one form and a knapsack weighing the other disagree by more than
|
||||||
|
the headroom. LF only, one trailing newline.
|
||||||
|
"""
|
||||||
|
return json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=False) + "\n"
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue