"""The consumption pre-pass, checked rather than described. `tools/okf_consume.py` cuts an OKF bundle to one contract-conformant payload for one question. Three disciplines this suite is held to, all of them the house pattern rather than new inventions: - **Every zero carries a control.** A count of nothing is a measurement whose query must first be shown capable of finding. The placeholder scan runs against the template (known-positive) before its zero on the filled copy is believed; the index walk is controlled against the `rglob` the contract forbids the consumer path from using; the socket guard is fired directly before its silence during a real run counts as evidence. - **One mutation per rule.** `tests/test_contract_check.py` establishes the shape: assert the code a defect produces, never merely that something failed. - **The corpus is never a test dependency.** K2 lives outside the repository. Every test here runs against `examples/.../expected-bundle` (3 concepts) or `tests/fixtures/consume-bundle` (synthetic, carrying the states the real corpus has zero of). The corpus-conditional arm skips with its denominator named, so a skip cannot read as a pass. """ from __future__ import annotations import hashlib import inspect import json import os import re import socket import subprocess import sys import unicodedata from collections.abc import Mapping from pathlib import Path from typing import Any import pytest PROJECT_ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(PROJECT_ROOT / "tools")) import okf_consume # noqa: E402 import okf_consume_measure # noqa: E402 import okf_contract_check # noqa: E402 import okf_skill # noqa: E402 from llm_ingestion_okf.materialize import parse_frontmatter # noqa: E402 TEMPLATE = PROJECT_ROOT / "skills" / "okf-consume-template" / "SKILL.md" def _skill_declaring(payload: dict[str, Any]) -> str: """A skill declaring the bundle THIS payload declares. The template cannot stand in for one any more: `` and `` are placeholders, and since 2026-09-10 an identity `okf check` cannot read is a `bundle_mismatch` finding. The sentence comes from the generator rather than being copied beside it.""" bundle = payload["bundle"] return TEMPLATE.read_text(encoding="utf-8").replace( okf_skill.TEMPLATE_HEADER, okf_skill.identity_line(bundle["bundle_id"], bundle["ref"]) + ".", 1, ) GOLDEN = PROJECT_ROOT / "examples" / "ingest-golden-segmented-okf-v0-2" / "expected-bundle" # --- Step 1: the walk and the ref --------------------------------------------- def test_the_index_walk_finds_every_concept_and_no_index_or_log() -> None: found = okf_consume.enumerate_concepts(GOLDEN) assert found == ( "krav/1-1/foerste-krav", "krav/1-2/andre-krav", "veiledning", ) assert not any(concept.endswith("index") or concept.endswith("log") for concept in found) def test_the_index_walk_is_complete_against_the_method_the_contract_forbids() -> None: # SS 9.2 forbids the CONSUMER path from enumerating a directory. Using the # forbidden method here, in a test, is what proves the permitted one loses # nothing -- an absence with no control is not a measurement. by_rglob = { path.relative_to(GOLDEN).with_suffix("").as_posix() for path in GOLDEN.rglob("*.md") if path.name not in ("index.md", "log.md") } assert by_rglob, "the control found nothing, so it cannot certify the walk" assert set(okf_consume.enumerate_concepts(GOLDEN)) == by_rglob def test_a_concept_reachable_only_through_a_nested_index_is_still_found() -> None: # `krav/1-1/foerste-krav` is three levels down and named in no root entry. root_entries = (GOLDEN / "index.md").read_text(encoding="utf-8") assert "foerste-krav" not in root_entries, "the fixture no longer exercises nesting" assert "krav/1-1/foerste-krav" in okf_consume.enumerate_concepts(GOLDEN) def test_the_ref_is_stable_across_calls_and_names_its_algorithm() -> None: first = okf_consume.bundle_ref(GOLDEN) assert first == okf_consume.bundle_ref(GOLDEN) assert first.startswith("sha256-tree:") def test_the_ref_moves_when_one_concept_byte_moves(tmp_path: Path) -> None: copy = tmp_path / "bundle" _copy_bundle(GOLDEN, copy) before = okf_consume.bundle_ref(copy) target = copy / "veiledning.md" target.write_text(target.read_text(encoding="utf-8") + "x", encoding="utf-8") assert okf_consume.bundle_ref(copy) != before def test_the_ref_does_not_move_when_only_mtimes_move(tmp_path: Path) -> None: copy = tmp_path / "bundle" _copy_bundle(GOLDEN, copy) before = okf_consume.bundle_ref(copy) for path in sorted(copy.rglob("*")): if path.is_file(): os.utime(path, (0, 0)) assert okf_consume.bundle_ref(copy) == before def _copy_bundle(source: Path, target: Path) -> None: for path in sorted(source.rglob("*")): if path.is_file(): destination = target / path.relative_to(source) destination.parent.mkdir(parents=True, exist_ok=True) destination.write_bytes(path.read_bytes()) def test_the_index_walk_excludes_a_linked_log_from_concept_navigation(tmp_path: Path) -> None: """A linked `log.md` is bundle metadata, not a concept -- measured on K2 (S7 F2). `link_log_in_root_index` (`corpus.py`, `95eb271`) linked a run's own log from the root index so a reader entering at `index.md` could reach it. That link makes the log reachable by the same walk this instrument uses to enumerate concepts, and a walk that does not distinguish "linked" from "concept" counts it as a 630th concept on a 629-concept bundle -- exactly what the S7 acid test measured, with the log then ranked and cut like real content. THE PRODUCER NO LONGER WRITES THAT LINK (2026-09-08), so the fixture writes it here instead. The exclusion stays and is not dead code: every bundle built between `95eb271` and that removal carries the link, including the ones consumers are reading today, and this instrument must count 629 on those too. """ from llm_ingestion_okf.corpus import LOG_NAME copy = tmp_path / "bundle" _copy_bundle(GOLDEN, copy) (copy / LOG_NAME).write_text("# Corpus run history\n\nN = 3\n", encoding="utf-8") index_path = copy / "index.md" index_path.write_text( index_path.read_text(encoding="utf-8") + f"- [Corpus run history]({LOG_NAME})\n", encoding="utf-8", newline="", ) index = index_path.read_text(encoding="utf-8") assert "](log.md)" in index, ( "the fixture must actually link the log for this control to mean anything" ) found = okf_consume.enumerate_concepts(copy) assert found == ( "krav/1-1/foerste-krav", "krav/1-2/andre-krav", "veiledning", ) assert not any(concept.endswith("log") for concept in found) def test_the_ref_covers_the_indexes_too_since_the_walk_reads_them(tmp_path: Path) -> None: # The docstring claims every byte that can reach a payload is inside the # ref. An index byte can: it decides which concepts are reachable at all. copy = tmp_path / "bundle" _copy_bundle(GOLDEN, copy) before = okf_consume.bundle_ref(copy) nested = copy / "krav" / "1-1" / "index.md" nested.write_text(nested.read_text(encoding="utf-8") + "\nfritekst\n", encoding="utf-8") assert okf_consume.bundle_ref(copy) != before # --- Step 2: one concept, read into a record ---------------------------------- PROPOSED_CONCEPT = "krav/1-1/foerste-krav" ROOT_BUNDLE_ID = "b-golden-segmented-okf-v0-2" def _read(concept_id: str, root: Path = GOLDEN) -> okf_consume.Concept: return okf_consume.read_concept( root / f"{concept_id}.md", bundle_root=root, root_bundle_id=ROOT_BUNDLE_ID ) def test_a_concept_carrying_adjudication_reads_that_value() -> None: assert _read(PROPOSED_CONCEPT).adjudication == "proposed" def test_a_concept_carrying_no_adjudication_key_reads_as_unknown(tmp_path: Path) -> None: # SS 6.1: `unknown` is written EXPLICITLY. "Not judged" and "we cannot tell # whether it was judged" are different facts, and only one is about the # concept. root = tmp_path / "bundle" _copy_bundle(GOLDEN, root) target = root / f"{PROPOSED_CONCEPT}.md" target.write_text( target.read_text(encoding="utf-8").replace("adjudication: proposed\n", ""), encoding="utf-8", ) concept = _read(PROPOSED_CONCEPT, root) assert concept.adjudication == "unknown" assert concept.adjudication_present is False def test_an_adjudication_value_outside_the_wire_set_is_refused_by_name(tmp_path: Path) -> None: # Mapping an unrecognised value to `unknown` would report "we cannot tell" # where the truth is "the bundle said something this consumer does not # understand" -- a defect laundered into a state. root = tmp_path / "bundle" _copy_bundle(GOLDEN, root) target = root / f"{PROPOSED_CONCEPT}.md" target.write_text( target.read_text(encoding="utf-8").replace("adjudication: proposed", "adjudication: seen"), encoding="utf-8", ) with pytest.raises(okf_consume.ConsumeError) as raised: _read(PROPOSED_CONCEPT, root) assert raised.value.code == "adjudication_unknown_value" def test_the_digest_is_of_the_concept_file_and_is_not_the_source_sha256() -> None: concept = _read(PROPOSED_CONCEPT) on_disk = hashlib.sha256((GOLDEN / f"{PROPOSED_CONCEPT}.md").read_bytes()).hexdigest() assert concept.sha256 == on_disk assert len(concept.sha256) == 64 frontmatter = parse_frontmatter(GOLDEN / f"{PROPOSED_CONCEPT}.md") assert frontmatter["source_sha256"] != concept.sha256 def test_the_concept_id_keeps_its_slashes_where_import_slug_would_flatten_them() -> None: # `importer.import_slug` flattens one line below the rule this id follows. # A flattened id fails a document-prefix match in a way that looks like a # ranking miss rather than an id-format bug. assert _read(PROPOSED_CONCEPT).concept_id == "krav/1-1/foerste-krav" def test_bundle_id_falls_back_to_the_root_index_and_says_that_it_did(tmp_path: Path) -> None: root = tmp_path / "bundle" _copy_bundle(GOLDEN, root) target = root / f"{PROPOSED_CONCEPT}.md" target.write_text( target.read_text(encoding="utf-8").replace(f"bundle_id: {ROOT_BUNDLE_ID}\n", ""), encoding="utf-8", ) concept = _read(PROPOSED_CONCEPT, root) assert concept.bundle_id == ROOT_BUNDLE_ID assert concept.bundle_id_inherited is True assert _read(PROPOSED_CONCEPT).bundle_id_inherited is False # --- Step 3: trust_tier, and the refusal to tier what cannot be read ---------- FIXTURE = PROJECT_ROOT / "tests" / "fixtures" / "consume-bundle" def test_the_fixture_bundle_carries_what_the_real_corpus_has_none_of() -> None: # The control on every assertion below. K2 has 0 `verified:` keys, 0 # `type: verdict` and 0 `adjudication: adjudicated` over 629 concepts, so a # fixture missing any of them would make its tests pass over an empty set. text = "\n".join(path.read_text(encoding="utf-8") for path in sorted(FIXTURE.rglob("*.md"))) assert "type: verdict" in text assert "adjudication: adjudicated" in text assert "by: human:" in text assert "by: process:" in text def test_no_verified_key_reads_as_unverified() -> None: # SS 6.3: a concept carrying no trust frontmatter is still consumable. assert okf_consume.trust_tier(None) == "unverified" def test_a_human_actor_reads_as_human_reviewed() -> None: assert okf_consume.trust_tier("[{ by: human:ktg, at: 2026-09-01T00:00:00Z }]") == ( "human-reviewed" ) def test_a_process_actor_reads_as_machine_confirmed() -> None: assert okf_consume.trust_tier("[{ by: process:okf-check, at: 2026-09-01T00:00:00Z }]") == ( "machine-confirmed" ) def test_the_human_test_is_a_prefix_and_never_a_substring() -> None: # `bot/human:2` is a MACHINE actor whose id contains the string `human:`. # A substring test would promote it to the highest tier -- fabricated # provenance produced by a matching bug. assert okf_consume.trust_tier("[{ by: bot/human:2 }]") == "machine-confirmed" def test_an_entry_naming_no_actor_is_refused_rather_than_tiered() -> None: with pytest.raises(okf_consume.ConsumeError) as raised: okf_consume.trust_tier("[{ at: 2026-09-01T00:00:00Z }]") assert raised.value.code == "verified_actorless" def test_a_block_form_verified_is_not_a_tier_at_all() -> None: # Measured 2026-09-07: this library's line-oriented `parse_frontmatter` # returns `''` for a block-form `verified:` and the full string for a flow # one, so PRESENT-BUT-UNREADABLE is distinguishable from ABSENT. Emitting # `unverified` here would assert a fact nobody measured (SS 6.4). assert okf_consume.trust_tier("") is None def test_the_block_form_case_is_real_in_the_fixture_and_not_only_in_the_unit_test() -> None: # The control: without this, `trust_tier("") is None` could be true of a # string no bundle ever produces. frontmatter = parse_frontmatter(FIXTURE / "dyp" / "nivaa" / "blokkform-verifisert.md") assert frontmatter["verified"] == "" assert okf_consume.trust_tier(frontmatter["verified"]) is None # --- Step 4: the budget instrument ------------------------------------------- CONTRACT = PROJECT_ROOT / "docs" / "consumption-contract.md" def test_measure_counts_bytes_and_not_characters() -> None: # The exact conflation the brief records itself making once: a chars/token # ratio quoted where a bytes/token one was needed. `æøå` is three # characters and six bytes, and the two only differ outside ASCII. assert okf_consume.measure("æøå") == len('"æøå"'.encode()) assert okf_consume.measure("æøå") != len("æøå") def test_measure_counts_the_encoded_form_the_payload_actually_costs() -> None: # `json.dumps` defaults to `ensure_ascii=True`, which inflates this corpus # by 7.1 %. A gate measuring one form while the knapsack weighs the other # disagrees by more than the headroom. norwegian = "årlig kontroll av anlegget" assert okf_consume.measure(norwegian) == len( json.dumps(norwegian, ensure_ascii=False).encode("utf-8") ) assert okf_consume.measure(norwegian) < len( json.dumps(norwegian, ensure_ascii=True).encode("utf-8") ) def test_the_known_positive_is_reproduced_by_the_gates_own_instrument() -> None: case, expected, measured = okf_consume.known_positive() assert case assert expected == measured, "SS 7.4: the instrument has not been shown to count" def test_the_known_positive_is_not_the_raw_byte_count_of_the_same_file() -> None: # Validating one instrument while gating with another is the SS 7.4 failure # the rule exists to prevent. The delta is derivable by a second, wholly # independent route (`wc -c`) and moves the moment `measure` changes what it # counts -- which is what keeps `expected == measured` from being vacuous. _, expected, _ = okf_consume.known_positive() raw = len(CONTRACT.read_bytes()) assert expected != raw assert expected - raw == okf_consume.KNOWN_POSITIVE_ENCODING_DELTA def test_the_default_limit_admits_a_concept_the_size_of_the_price_form() -> None: # Measured during planning: at the drafted 60 000 B default the SC6 gold # concept (101 313 B encoded) falls to the "cannot fit alone" pre-exclusion, # so SC1 and SC6 were mutually unsatisfiable on a CORRECT implementation. assert okf_consume.DEFAULT_LIMIT >= 101_313 def test_the_budget_unit_and_instrument_are_named_rather_than_implied() -> None: assert "byte" in okf_consume.BUDGET_UNIT assert "ensure_ascii=False" in okf_consume.BUDGET_INSTRUMENT # --- Step 5: stage-one document ranking -------------------------------------- def test_normalise_is_nfc_stable_on_the_one_letter_that_decomposes() -> None: # `NFD("å")` is `a` + U+030A, and the combining ring is not `\w`, so an # un-normalised split returns `["a", "rlig"]`. `æ` and `ø` have NO canonical # decomposition, so a test built on `miljø` passes while the bug is live -- # the known-positive here MUST use `å`. composed = unicodedata.normalize("NFC", "årlig kontroll") decomposed = unicodedata.normalize("NFD", "årlig kontroll") assert composed != decomposed, "the control is broken: the two forms are identical" assert okf_consume.normalise(decomposed) == okf_consume.normalise(composed) assert "årlig" in okf_consume.normalise(decomposed) def test_normalise_drops_tokens_under_three_characters() -> None: assert okf_consume.normalise("er en pris i et skjema") == ("pris", "skjema") def test_normalise_holds_an_identifier_number_as_one_token() -> None: # MEASURED 2026-09-08 over three vegnormal bundles (446, 1133 and 270 # concepts): `_TOKEN_SPLIT_RE` shatters `10.2-2` into `10`, `2`, `2` and # `MIN_TOKEN_LENGTH` then drops every piece, so a question naming a # requirement number reaches the ranker carrying only the word `krav` -- # which every concept in such a bundle also carries. BOTH mechanisms # participate: the split destroys the number, the floor removes the # remains. The gold requirement was `below_k` in three of three. assert "10.2-2" in okf_consume.normalise("Krav 10.2\u20142") assert okf_consume.normalise("3.3.1\u201413") == ("3.3.1-13",) assert okf_consume.normalise("2.9.2\u201412") == ("2.9.2-12",) assert okf_consume.normalise("R610.4") == ("r610.4",) assert okf_consume.normalise("4.2.1") == ("4.2.1",) def test_an_identifiers_three_spellings_of_its_separator_normalise_alike() -> None: # One requirement number arrives as an em dash from the source viewer, an # en dash from a converter and a plain hyphen from a person typing the # question. NFC folds NONE of the three, so a rule that does not fold them # finds the number only in the spelling it was asked with. hyphen = okf_consume.normalise("10.2-2") assert okf_consume.normalise("10.2\u20142") == hyphen assert okf_consume.normalise("10.2\u20132") == hyphen assert hyphen == ("10.2-2",) def test_the_identifier_rule_leaves_the_noise_floor_it_was_added_under() -> None: # The known-negative, and the whole reason `MIN_TOKEN_LENGTH` exists: a # bare short number matches every page number, row count and year in a # corpus, and a matcher that scores them ranks every document equally. assert okf_consume.normalise("10") == () assert okf_consume.normalise("2") == () assert okf_consume.normalise("er en pris i et skjema") == ("pris", "skjema") # A hyphenated WORD is not an identifier -- no digit stands on either side # of the separator -- so it splits exactly as it always did. assert okf_consume.normalise("skole-anbudet") == ("skole", "anbudet") # And a separator this rule does not claim leaves its token set untouched: # the K2 corpus spells standards this way. assert okf_consume.normalise("NS3935:2019") == ("ns3935", "2019") assert okf_consume.normalise("TEK 17") == ("tek",) def test_an_identifier_inside_a_slug_does_not_swallow_the_words_around_it() -> None: # MEASURED, and the reason the rule joins DIGIT groups rather than # alphanumeric ones. A first version joined alphanumeric groups across a # separator; a corpus document's slug then became ONE token, because a # `3-6` sits inside it, and that document's score for a question naming its # subject fell from 0.735 to 0.0 -- one hit@8 row lost, on a question # carrying no identifier at all. The slug below has that shape and is not # the corpus's (SS "consumer content stays at form level"). The identifier # is ADDED here; nothing is taken away. tokens = okf_consume.normalise("rapport-iv-vedlegg-3-6-grunnforhold-akustikk") assert "akustikk" in tokens assert "grunnforhold" in tokens assert "vedlegg" in tokens assert "3-6" in tokens def test_two_tokens_match_on_a_shared_prefix_of_four_and_not_of_three() -> None: # "Stem-substring" is not an implementable rule: neither `prisene` nor # `prissammenstilling` contains the other. Shared prefix does the work -- # `pris|ene` and `pris|sammenstilling` share 4. A 3-character floor # over-matches Norwegian function words. assert okf_consume.tokens_match("prisene", "prissammenstilling") assert okf_consume.tokens_match("kontrollen", "kontroll") assert not okf_consume.tokens_match("pris", "pri") assert not okf_consume.tokens_match("krav", "kraft") def test_a_question_naming_a_directorys_subject_ranks_that_directory_first() -> None: scores = okf_consume.document_scores(FIXTURE, "Hvordan skal prisene fylles ut?") assert scores, "no document scored, so 'ranks first' would measure nothing" assert max(scores, key=lambda key: (scores[key], key)) == "krav" def test_a_question_about_a_different_subject_ranks_a_different_directory() -> None: # The control on the test above: without it, a scorer returning "krav" # unconditionally would pass. scores = okf_consume.document_scores(FIXTURE, "Hva er omfanget og formaalet?") assert max(scores, key=lambda key: (scores[key], key)) == "scope" def test_curated_prose_in_an_index_is_ignored_rather_than_scored(tmp_path: Path) -> None: root = tmp_path / "bundle" _copy_bundle(FIXTURE, root) index = root / "krav" / "index.md" index.write_text( "Denne mappen handler om priser og prissammenstilling.\n\n" + index.read_text(encoding="utf-8"), encoding="utf-8", ) assert ( okf_consume.DEFAULT_PROFILE.index.parse_entry( "Denne mappen handler om priser og prissammenstilling." ) is None ) assert okf_consume.document_scores(root, "Hvordan skal prisene fylles ut?") == ( okf_consume.document_scores(FIXTURE, "Hvordan skal prisene fylles ut?") ) def test_document_scores_are_identical_across_two_calls() -> None: question = "Hvordan skal prisene fylles ut?" assert okf_consume.document_scores(FIXTURE, question) == okf_consume.document_scores( FIXTURE, question ) # --- Step 6: stage-two concept ranking, fused by RRF -------------------------- def _fixture_concepts() -> list[okf_consume.Concept]: return [ okf_consume.read_concept( FIXTURE / f"{concept_id}.md", bundle_root=FIXTURE, root_bundle_id="consume-fixture", ) for concept_id in okf_consume.enumerate_concepts(FIXTURE) ] def test_concepts_tying_on_every_signal_come_back_in_concept_id_order() -> None: concepts = _fixture_concepts() # A question matching nothing makes every signal identical, so the ONLY # thing left deciding the order is the declared tie-break. ranked = okf_consume.concept_scores(concepts, "zzzz qqqq", {}) ids = [concept.concept_id for concept, _, _ in ranked] assert ids == sorted(ids) def test_reversing_the_input_order_does_not_change_the_output_order() -> None: concepts = _fixture_concepts() forward = [c.concept_id for c, _, _ in okf_consume.concept_scores(concepts, "zzzz qqqq", {})] backward = [ c.concept_id for c, _, _ in okf_consume.concept_scores(list(reversed(concepts)), "zzzz qqqq", {}) ] assert forward == backward def test_a_concept_in_a_high_scoring_document_outranks_an_equally_lexical_one() -> None: # `tie_shared_rank=False` for the same reason `lookup=False` appears # elsewhere in this file: the claim is about the DOCUMENT PRIOR, and the # default tie-break (shared since 2026-09-10) puts this fixture's two # concepts in the same prior tie group, which makes the assertion true # in both directions and so measures nothing. Isolate the stage under test. concepts = _fixture_concepts() question = "Hvordan skal prisene fylles ut?" lifted = okf_consume.concept_scores( concepts, question, {"krav": 10.0, "dyp": 0.0}, tie_shared_rank=False ) dropped = okf_consume.concept_scores( concepts, question, {"krav": 0.0, "dyp": 10.0}, tie_shared_rank=False ) krav_first = [c.concept_id for c, _, _ in lifted].index("krav/pristabell") krav_later = [c.concept_id for c, _, _ in dropped].index("krav/pristabell") assert krav_first < krav_later def test_the_ranked_order_is_identical_across_two_calls() -> None: concepts = _fixture_concepts() scores = okf_consume.document_scores(FIXTURE, "Hvordan skal prisene fylles ut?") first = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores) second = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores) assert [c.concept_id for c, _, _ in first] == [c.concept_id for c, _, _ in second] def test_the_price_concept_leads_on_the_price_question_in_the_fixture() -> None: concepts = _fixture_concepts() scores = okf_consume.document_scores(FIXTURE, "Hvordan skal prisene fylles ut?") ranked = okf_consume.concept_scores(concepts, "Hvordan skal prisene fylles ut?", scores) assert ranked[0][0].concept_id == "krav/pristabell" # --- Step 7: the cut ---------------------------------------------------------- def _cut_fixture( question: str = "Hvordan skal prisene fylles ut?", k: int = 8, limit: int | None = None ) -> tuple[list[dict[str, object]], list[tuple[str, str]], int]: concepts = _fixture_concepts() scores = okf_consume.document_scores(FIXTURE, question) ranked = okf_consume.concept_scores(concepts, question, scores) delivered, withheld, _ = okf_consume.cut( ranked, k=k, limit=okf_consume.DEFAULT_LIMIT if limit is None else limit ) return list(delivered), list(withheld), len(ranked) def test_the_fixture_has_both_a_verdict_concept_and_a_delivered_one() -> None: # The control on every count below: neither zero may come from an empty # fixture. delivered, withheld, considered = _cut_fixture() assert delivered, "nothing was delivered, so 'excluded' would measure nothing" assert withheld, "nothing was withheld, so the rules would measure nothing" assert considered == len(okf_consume.enumerate_concepts(FIXTURE)) def test_a_verdict_concept_is_withheld_by_rule_and_reaches_no_excerpt() -> None: delivered, withheld, _ = _cut_fixture() rules = dict(withheld) assert rules["dyp/nivaa/alminnelig-notat"] == "verdict_layer_excluded" assert all(excerpt["concept_id"] != "dyp/nivaa/alminnelig-notat" for excerpt in delivered) def test_the_verdict_exclusion_is_a_type_check_and_never_a_path_filter() -> None: # One question reaching BOTH: a `type: reference` file NAMED # `verdict-lookalike` is delivered, and a `type: verdict` file under an # ordinary name at depth 3 is withheld. A path filter gets both backwards, # and neither half of this is measured unless both concepts match. delivered, withheld, _ = _cut_fixture(question="Hva sier notatet om stifilter og typesjekk?") delivered_ids = {excerpt["concept_id"] for excerpt in delivered} rules = dict(withheld) assert "krav/verdict-lookalike" in delivered_ids assert "dyp/nivaa/alminnelig-notat" not in delivered_ids assert rules["dyp/nivaa/alminnelig-notat"] == "verdict_layer_excluded" assert rules["dyp/nivaa/alminnelig-notat"] != "no_lexical_match" def test_a_capital_l_log_type_does_not_crash_the_reader() -> None: # `type: Log` really occurs in the K2 corpus. Case handling is a test here # rather than an accident. delivered, withheld, _ = _cut_fixture() seen = {excerpt["concept_id"] for excerpt in delivered} | {cid for cid, _ in withheld} assert "krav/loggnotat" in seen def test_a_concept_whose_verified_cannot_be_read_is_withheld_by_name() -> None: # SS 6.2 requires a tier on every excerpt and SS 6.4 forbids reading absence # as negation. Emitting `unverified` for an unreadable value asserts a fact # nobody measured. # The question must REACH the concept: a relevance drop fires first, and a # `no_lexical_match` here would prove nothing about tiering. _, withheld, _ = _cut_fixture(question="Hva staar i blokkform?") assert dict(withheld)["dyp/nivaa/blokkform-verifisert"] == "verified_unreadable" def test_a_withheld_entry_names_what_was_dropped_under_the_flag() -> None: # A reader who is told 262 concepts were withheld, by id and rule alone, # cannot tell WHAT was withheld without reading the bundle -- which SS 2.2 # forbids. The title closes that, and it is emitted only where the concept # carries one. payload = okf_consume.build_payload( FIXTURE, question="Hvordan skal prisene fylles ut?", withheld_titles=True ) entries = payload["withheld"] assert isinstance(entries, list) and entries titled = [entry for entry in entries if "title" in entry] assert titled, "no withheld entry carried a title, so the rule measures nothing" concepts = {concept.concept_id: concept for concept in _fixture_concepts()} for entry in entries: concept = concepts[str(entry["concept_id"])] if concept.title: assert entry["title"] == concept.title else: assert "title" not in entry def test_no_withheld_entry_names_anything_without_the_flag() -> None: # The default is what every consumer already runs, and this is the # measurement that keeps it theirs: a title on every withheld entry grew a # 270-concept payload by 37.9 % and pushed a 629-concept bundle's # bookkeeping past the budget limit itself. payload = okf_consume.build_payload(FIXTURE, question="Hvordan skal prisene fylles ut?") entries = payload["withheld"] assert isinstance(entries, list) and entries assert all(set(entry) == {"concept_id", "rule"} for entry in entries) def test_the_withheld_title_flag_costs_bytes_and_the_default_pays_none() -> None: question = "Hvordan skal prisene fylles ut?" off = okf_consume.serialise(okf_consume.build_payload(FIXTURE, question=question)) explicit_off = okf_consume.serialise( okf_consume.build_payload(FIXTURE, question=question, withheld_titles=False) ) on = okf_consume.serialise( okf_consume.build_payload(FIXTURE, question=question, withheld_titles=True) ) assert off == explicit_off assert len(on.encode("utf-8")) > len(off.encode("utf-8")) def test_delivered_and_withheld_partition_the_considered_set() -> None: delivered, withheld, considered = _cut_fixture() delivered_ids = {excerpt["concept_id"] for excerpt in delivered} withheld_ids = {concept_id for concept_id, _ in withheld} assert delivered_ids & withheld_ids == set() assert len(delivered_ids) + len(withheld_ids) == considered assert delivered_ids | withheld_ids == set(okf_consume.enumerate_concepts(FIXTURE)) def test_every_withheld_entry_names_a_rule_from_the_closed_set() -> None: _, withheld, _ = _cut_fixture() assert {rule for _, rule in withheld} <= set(okf_consume.WITHHOLDING_RULES) def test_a_concept_larger_than_the_limit_is_excluded_by_name_before_the_dp() -> None: # Named as a RULE rather than left as a packing artefact: "it did not fit" # and "it could never fit" are different facts about the cut. _, withheld, _ = _cut_fixture(limit=200) rules = {rule for _, rule in withheld} assert "over_budget_alone" in rules assert "over_budget_after_knapsack" not in rules def test_concepts_ranked_beyond_k_are_withheld_as_below_k() -> None: # A question matching TWO concepts, so that capping at one leaves a real # `below_k` drop rather than an empty set. matching = "kontroll av prisene" _, wide, _ = _cut_fixture(question=matching, k=8) assert "below_k" not in {rule for _, rule in wide} _, narrow, _ = _cut_fixture(question=matching, k=1) assert "below_k" in {rule for _, rule in narrow} def test_the_exact_knapsack_beats_greedy_by_density() -> None: # Greedy takes the densest item first and is then unable to fit either of # the two that together are worth more. Greedy-by-density has an unbounded # approximation factor; an exact DP over at most `k` items is microseconds. items = ((10.0, 6), (7.0, 5), (7.0, 5)) chosen = okf_consume.knapsack(items, capacity=10) assert sorted(chosen) == [1, 2] assert sum(items[index][0] for index in chosen) == 14.0 def test_the_knapsack_is_deterministic_over_equal_value_subsets() -> None: items = ((5.0, 5), (5.0, 5), (5.0, 5)) assert okf_consume.knapsack(items, capacity=10) == okf_consume.knapsack(items, capacity=10) def test_excerpts_come_back_in_rank_order_and_carry_that_rank() -> None: # An id-sorted payload would turn "position in the payload" into a # different number from "position in the ranking", and hit@k reads the # second one. delivered, _, _ = _cut_fixture() assert [excerpt["rank"] for excerpt in delivered] == list(range(1, len(delivered) + 1)) assert delivered[0]["concept_id"] == "krav/pristabell" # --- Step 8: the payload ------------------------------------------------------ def _payload( root: Path = FIXTURE, question: str = "Hvordan skal prisene fylles ut?", **kwargs: object ) -> dict[str, object]: return okf_consume.build_payload(root, question=question, **kwargs) # type: ignore[arg-type] def test_the_payload_passes_the_checker_against_a_skill_for_its_own_bundle() -> None: payload = _payload() report = okf_contract_check.check(_skill_declaring(payload), payload) assert report.findings == () def test_the_payload_carries_every_section_eight_member() -> None: payload = _payload() assert payload["contract"] == "okf-consumption/1" assert set(payload) >= { "contract", "bundle", "budget", "denominators", "excerpts", "withheld", } bundle = payload["bundle"] assert isinstance(bundle, dict) assert bundle["bundle_id"] == "consume-fixture" assert str(bundle["ref"]).startswith("sha256-tree:") def test_spent_is_the_cost_of_the_delivered_set_and_not_of_the_whole_payload() -> None: # SS 7.2 verbatim: "what the DELIVERED SET spent by that same instrument". # Measured on K2 at k=8, the whole-payload reading puts a 628-entry # `withheld` list (81 565 B) plus one gold excerpt (101 576 B) against a # 120 000 B limit -- so a CORRECT implementation would exit 1 and fail its # own SC1 and SC6. payload = _payload() budget = payload["budget"] excerpts = payload["excerpts"] assert isinstance(budget, dict) and isinstance(excerpts, list) assert budget["spent"] == sum(okf_consume.excerpt_weight(e) for e in excerpts) def test_spent_moves_when_an_excerpt_moves_and_holds_when_withheld_grows() -> None: # The property that distinguishes SS 7.2's reading from the whole-payload # one, asserted rather than described. matching = "kontroll av prisene" wide = _payload(question=matching, k=8) narrow = _payload(question=matching, k=1) wide_budget, narrow_budget = wide["budget"], narrow["budget"] wide_counts, narrow_counts = wide["denominators"], narrow["denominators"] assert isinstance(wide_budget, dict) and isinstance(narrow_budget, dict) assert isinstance(wide_counts, dict) and isinstance(narrow_counts, dict) assert narrow_counts["withheld"] > wide_counts["withheld"] assert narrow_budget["spent"] < wide_budget["spent"] def test_the_counts_and_the_lists_are_two_statements_of_one_fact() -> None: payload = _payload() counts, excerpts, withheld = payload["denominators"], payload["excerpts"], payload["withheld"] assert isinstance(counts, dict) and isinstance(excerpts, list) and isinstance(withheld, list) assert counts["delivered"] == len(excerpts) assert counts["withheld"] == len(withheld) assert counts["considered"] == counts["delivered"] + counts["withheld"] def test_every_excerpt_digest_recomputes_from_the_named_concepts_bytes() -> None: payload = _payload() excerpts = payload["excerpts"] assert isinstance(excerpts, list) and excerpts for excerpt in excerpts: on_disk = FIXTURE / f"{excerpt['concept_id']}.md" assert excerpt["sha256"] == hashlib.sha256(on_disk.read_bytes()).hexdigest() assert ( excerpt["text_sha256"] == hashlib.sha256(str(excerpt["text"]).encode("utf-8")).hexdigest() ) def test_every_concept_id_keeps_the_slash_import_slug_would_have_removed() -> None: payload = _payload() excerpts = payload["excerpts"] assert isinstance(excerpts, list) assert any("/" in str(excerpt["concept_id"]) for excerpt in excerpts) def test_the_budget_refuses_rather_than_narrowing_when_the_cut_cannot_fit() -> None: # SS 7.3: exceeding the gate is a finding requiring a decision, never # something to retry narrower. with pytest.raises(okf_consume.ConsumeError) as raised: _payload(limit=100) assert raised.value.code == "budget_admits_nothing" def test_the_serialised_payload_is_lf_only_and_ends_in_exactly_one_newline() -> None: text = okf_consume.serialise(_payload()) assert "\r" not in text assert text.endswith("}\n") assert not text.endswith("}\n\n") def test_the_serialised_payload_does_not_escape_norwegian_letters() -> None: # `ensure_ascii=True` inflates this corpus by 7.1 %, which is more than the # headroom the gate leaves. text = okf_consume.serialise(_payload(question="Hvor ofte er den årlige kontrollen?")) assert "\\u00e5" not in text def test_a_question_with_no_answer_returns_a_measured_empty_set_not_a_guess() -> None: # The order's known-negative control, and the reason the `no_lexical_match` # rule exists: a ranker that always returns its top eight scores well on # every positive question and is useless. The emptiness must be POSITIVE -- # every considered concept named in `withheld` under a rule, so the identity # still closes and the skill can say "measured, nothing cleared the bar" # rather than "nothing was found". payload = _payload(question="Hva er reglene for sveising av titan i vakuum?") counts, excerpts, withheld = payload["denominators"], payload["excerpts"], payload["withheld"] assert isinstance(counts, dict) and isinstance(excerpts, list) and isinstance(withheld, list) assert excerpts == [] assert counts["delivered"] == 0 assert ( counts["withheld"] == counts["considered"] == len(okf_consume.enumerate_concepts(FIXTURE)) ) assert {entry["rule"] for entry in withheld} == {"no_lexical_match", "verdict_layer_excluded"} # And the control: the SAME payload builder returns a non-empty set for a # question this bundle does answer, so the zero is a measurement. answered = _payload() answered_excerpts = answered["excerpts"] assert isinstance(answered_excerpts, list) and answered_excerpts def test_the_empty_payload_still_passes_the_checker() -> None: payload = _payload(question="Hva er reglene for sveising av titan i vakuum?") assert okf_contract_check.check(_skill_declaring(payload), payload).findings == () # --- Corpus-conditional arms -------------------------------------------------- K2_BUNDLE = Path.home() / "corpora" / "okf-telling-20260829" / "K2-bundle-20260903" K2_CONCEPTS = 629 K2_PROPOSED = 618 K2_KEYLESS = 11 requires_k2 = pytest.mark.skipif( not K2_BUNDLE.is_dir(), reason=( f"the K2 corpus is not present at {K2_BUNDLE}. NOT MEASURED, not zero: " f"this arm covers a denominator of {K2_CONCEPTS} concepts, of which " f"{K2_PROPOSED} carry `adjudication: proposed` and {K2_KEYLESS} carry no " "`adjudication` key at all. A skip here is an unmeasured denominator, " "never a pass." ), ) @requires_k2 def test_the_eleven_keyless_k2_concepts_come_back_unknown_over_a_stated_denominator() -> None: # SS 6.1's third state, on real data rather than on a fixture. The 11 are # asserted as ONE named set: measured, the concepts carrying no # `adjudication` are EXACTLY those carrying no `bundle_id`, so three # independent counts would share one blind spot. root_bundle_id = parse_frontmatter(K2_BUNDLE / "index.md")["bundle_id"] concepts = [ okf_consume.read_concept( K2_BUNDLE / f"{concept_id}.md", bundle_root=K2_BUNDLE, root_bundle_id=root_bundle_id, ) for concept_id in okf_consume.enumerate_concepts(K2_BUNDLE) ] assert len(concepts) == K2_CONCEPTS unknown = {c.concept_id for c in concepts if c.adjudication == "unknown"} inherited = {c.concept_id for c in concepts if c.bundle_id_inherited} proposed = [c for c in concepts if c.adjudication == "proposed"] assert len(proposed) == K2_PROPOSED assert len(unknown) == K2_KEYLESS assert unknown == inherited, "the two sets diverged; the fallback is no longer one fact" assert all(c.bundle_id == root_bundle_id for c in concepts if c.bundle_id_inherited) # `adjudicated` has denominator ZERO on this corpus. Stated, not implied. assert [c for c in concepts if c.adjudication == "adjudicated"] == [] @requires_k2 def test_spent_is_the_delivered_set_where_the_whole_payload_reading_would_refuse() -> None: # The regression guard, with figures RE-MEASURED here rather than carried # from the plan: the plan predicted 101 576 B for this excerpt and 188 758 B # for the payload, both taken before per-line trailing-whitespace stripping # landed. What this build actually produces is recorded instead. payload = okf_consume.build_payload(K2_BUNDLE, question="Hvordan skal prisene fylles ut?") budget, excerpts = payload["budget"], payload["excerpts"] assert isinstance(budget, dict) and isinstance(excerpts, list) whole_payload = len(okf_consume.serialise(payload).encode("utf-8")) assert whole_payload > int(budget["limit"]), ( "the guard measures nothing: the whole payload already fits, so the two " "readings of SS 7.2 cannot be told apart on this case" ) assert int(budget["spent"]) <= int(budget["limit"]) #: The gold set is LOCAL-ONLY: it names corpus documents, which never reach a #: tracked file here. The test reads it rather than restating it, so this file #: carries the assertion and not the answer key. GOLD_SET = PROJECT_ROOT / ".claude/projects/2026-09-07-okf-consume-prepass/hit-at-k-questions.json" @requires_k2 @pytest.mark.skipif(not GOLD_SET.is_file(), reason=f"the local gold set is absent ({GOLD_SET})") def test_every_gold_document_in_the_local_set_is_reached_or_named_as_a_miss() -> None: # SC5 and SC6 together, run against the answer key rather than a literal. # Row 1's gold is the one confirmed by a signal from outside this # repository -- a live model reached that document unprompted in three # navigation steps on 2026-09-06 -- and its gold document holds exactly one # concept, so it is also the one concept-granularity row. spec = json.loads(GOLD_SET.read_text(encoding="utf-8")) questions = spec["questions"] assert len(questions) >= 5, "fewer than five questions is not the measurement" hits = 0 for entry in questions: payload = okf_consume.build_payload(K2_BUNDLE, question=entry["question"]) excerpts = payload["excerpts"] assert isinstance(excerpts, list) if okf_consume_measure.hit_rank(excerpts, entry["gold_document"]) is not None: hits += 1 # The published bar, and the published number. A regression that drops a # row goes red here rather than in a document nobody re-runs. # # 5 -> 6 ON 2026-09-10, with no bundle changing: `DEFAULT_SOURCE_QUOTA = 2` # reaches the one row that had missed everywhere. What that gain is not: # this metric asks whether the gold DOCUMENT was delivered, and a document # quota raises how many distinct documents a payload holds, so it is not # neutral with respect to the rule that moved it. assert hits == 6, f"hit@8 moved: {hits} of {len(questions)}" # --- Step 9: the CLI ---------------------------------------------------------- TOOL = PROJECT_ROOT / "tools" / "okf_consume.py" def _run(*args: str) -> subprocess.CompletedProcess[str]: return subprocess.run( [sys.executable, str(TOOL), *args], capture_output=True, text=True, check=False ) def test_two_runs_of_the_same_arguments_produce_byte_identical_stdout() -> None: first = _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?") second = _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?") assert first.returncode == 0, first.stderr assert first.stdout == second.stdout assert first.stdout def test_the_module_reaches_no_clock() -> None: # Determinism is a property of the code, not only of two runs that happened # to land in the same second. source = TOOL.read_text(encoding="utf-8") for forbidden in ("datetime.now", "time.time", "utcnow", "time.monotonic"): assert forbidden not in source def test_exit_zero_one_and_two_are_each_reached_by_a_distinct_real_condition() -> None: ok = _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?") assert ok.returncode == 0 refused = _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?", "--limit", "100") assert refused.returncode == 1 assert "budget" in refused.stderr absent = _run(str(FIXTURE / "does-not-exist"), "--question", "Hva som helst her") assert absent.returncode == 2 def test_a_matching_ref_passes_and_a_mismatching_one_refuses_and_writes_nothing( tmp_path: Path, ) -> None: # SS 3.3: `--ref` is an ASSERTION. An override would let a caller label a # payload with an identity its bytes do not have, which is the one thing # that paragraph exists to prevent. real = okf_consume.bundle_ref(FIXTURE) out = tmp_path / "payload.json" good = _run( str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?", "--ref", real, "--out", str(out), ) assert good.returncode == 0 assert json.loads(out.read_text(encoding="utf-8"))["bundle"]["ref"] == real missing = tmp_path / "never-written.json" bad = _run( str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?", "--ref", "sha256-tree:0000", "--out", str(missing), ) assert bad.returncode == 1 assert not missing.exists() def test_out_writes_exactly_the_bytes_stdout_produced(tmp_path: Path) -> None: out = tmp_path / "payload.json" piped = _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?") written = _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?", "--out", str(out)) assert written.returncode == 0 assert out.read_text(encoding="utf-8") == piped.stdout def test_every_top_level_import_is_stdlib_or_this_repository() -> None: # SC4, narrowed with the measurement that forced it: importing any library # primitive pulls `socket`/`ssl`/`urllib` transitively, because Door A # legitimately needs them. Reachability is not use. The honest guarantee is # no THIRD-PARTY dependency plus no network call, and the second half is # asserted below. source = TOOL.read_text(encoding="utf-8") imported = set(re.findall(r"^(?:from|import) ([a-zA-Z_][\w.]*)", source, re.MULTILINE)) for module in imported: root = module.split(".")[0] assert root in sys.stdlib_module_names or root == "llm_ingestion_okf", root def test_no_socket_is_opened_during_a_real_run(monkeypatch: pytest.MonkeyPatch) -> None: calls: list[object] = [] def refuse(*args: object, **kwargs: object) -> None: calls.append(args) raise AssertionError("the pre-pass opened a socket") monkeypatch.setattr(socket, "socket", refuse) monkeypatch.setattr(socket, "create_connection", refuse) # The guard proven able to fire, before its silence counts as evidence. with pytest.raises(AssertionError): socket.socket() calls.clear() okf_consume.build_payload(FIXTURE, question="Hvordan skal prisene fylles ut?") assert calls == [] def test_the_payload_written_by_the_cli_passes_the_checker(tmp_path: Path) -> None: out = tmp_path / "payload.json" assert ( _run(str(FIXTURE), "--question", "Hvordan skal prisene fylles ut?", "--out", str(out)) ).returncode == 0 skill_file = tmp_path / "SKILL.md" skill_file.write_text( _skill_declaring(json.loads(out.read_text(encoding="utf-8"))), encoding="utf-8" ) checked = subprocess.run( [ sys.executable, str(PROJECT_ROOT / "tools" / "okf_contract_check.py"), "--skill", str(skill_file), "--payload", str(out), ], capture_output=True, text=True, check=False, ) assert checked.returncode == 0, checked.stdout assert "0 findings" in checked.stdout # --- Step 10: the instantiated skill ----------------------------------------- SKILL = PROJECT_ROOT / "skills" / "okf-consume" / "SKILL.md" PLACEHOLDER_RE = re.compile(r"<[A-Z][A-Z_]{2,}(?::.*?)?>", re.DOTALL) def test_the_placeholder_scan_finds_them_in_the_template_before_its_zero_counts() -> None: # The known-positive, run FIRST. The obvious check is blind: a # line-oriented `<[A-Z_]*>` cannot match ``, # `` or ``, each of which spans # lines. Measured: the naive pattern reports 17 against 20 real occurrences. template = TEMPLATE.read_text(encoding="utf-8") naive = re.findall(r"<[A-Z_]*>", template) thorough = PLACEHOLDER_RE.findall(template) assert len(thorough) >= 20 assert len(thorough) > len(naive), "the scan is no better than the blind one" def test_the_instantiated_skill_has_no_placeholder_left() -> None: assert PLACEHOLDER_RE.findall(SKILL.read_text(encoding="utf-8")) == [] def test_the_instantiated_skill_carries_every_required_section() -> None: text = SKILL.read_text(encoding="utf-8") for section in okf_contract_check.REQUIRED_SECTIONS: assert f"## {section}" in text def test_the_instantiated_skill_carries_every_marking_and_state_literal() -> None: text = SKILL.read_text(encoding="utf-8") for marking in okf_contract_check.REQUIRED_MARKINGS: assert marking in text, marking for state in (*okf_contract_check.ADJUDICATION_STATES, *okf_contract_check.TRUST_TIERS): assert f"`{state}`" in text, state def test_every_rule_the_pre_pass_can_emit_is_named_in_the_skill() -> None: # The anti-drift gate. A rule the pre-pass emits and the skill does not # explain is a `withheld` entry no reader can act on, and the copy nobody # reads is the one that goes wrong. text = SKILL.read_text(encoding="utf-8") for rule in okf_consume.WITHHOLDING_RULES: assert rule in text, rule def test_the_skill_and_a_real_payload_pass_the_checker_together(tmp_path: Path) -> None: # A GENERATED skill, against a payload from the bundle it was generated for. # The shipped `skills/okf-consume/SKILL.md` cannot serve here: it predates # `okf skill` and declares no bundle identity a reader can act on, which is # a `bundle_mismatch` finding and is recorded as one rather than worked # around. text, payload = okf_skill.render(GOLDEN, out=tmp_path / "skill") assert okf_contract_check.check(text, payload).findings == () def test_the_shipped_example_payload_is_current_and_regenerates_byte_for_byte() -> None: # A shipped artefact that has drifted from the tool that made it is worse # than none: it documents a shape the code no longer emits. shipped = (SKILL.parent / "references" / "example-payload.json").read_text(encoding="utf-8") regenerated = okf_consume.serialise( okf_consume.build_payload(GOLDEN, question="Hva sier veiledningen om krav?") ) assert shipped == regenerated def test_the_shipped_skill_passes_the_checker_against_its_own_payload() -> None: # The pair this repository ships, read from disk and checked as it stands. # The hand-filled copy that stood here until 2026-09-11 was refused against # the payload beside it: "the skill declares no readable bundle identity". # A skill that fails the check it tells its reader to run is the one # artefact here that must not. text = SKILL.read_text(encoding="utf-8") payload = json.loads( (SKILL.parent / "references" / "example-payload.json").read_text(encoding="utf-8") ) report = okf_contract_check.check(text, payload) assert report.findings == (), report.render() assert report.rules_evaluated == len(okf_contract_check.RULES) def test_the_shipped_skill_is_the_generator_output_with_the_checkout_made_relative() -> None: # `okf skill` writes the bundle root and its own path ABSOLUTE when --out is # not under `.claude/skills/`, and `skills/okf-consume` is not. Shipped as # generated, the file would name one checkout by absolute path and its two # commands would run on one machine only. The single step after the # generator strips the checkout prefix, and this test is that step, so the # shipped bytes are the generator's bytes and nothing else. generated, _ = okf_skill.render( GOLDEN, out=SKILL.parent, question="Hva sier veiledningen om krav?" ) prefix = f"{PROJECT_ROOT}/" # The known-positive: the strip has something to strip, so the equality # below is not a comparison of two texts that never carried the prefix. assert prefix in generated assert SKILL.read_text(encoding="utf-8") == generated.replace(prefix, "") @requires_k2 def test_no_corpus_document_name_reaches_any_file_this_work_tracks() -> None: # CLAUDE.md's public-file rule. The pattern is DERIVED from the corpus's own # top-level document names at run time rather than hand-picked, so it covers # every document rather than the six someone thought of -- and so this # tracked file carries no corpus name of its own. documents = sorted( {concept_id.split("/", 1)[0] for concept_id in okf_consume.enumerate_concepts(K2_BUNDLE)} ) assert len(documents) > 30, "too few documents to be the real corpus" leak = re.compile("|".join(re.escape(name) for name in documents), re.IGNORECASE) # The known-positive, first: the pattern must be shown able to find before # its zero counts as a measurement. control = (K2_BUNDLE / "index.md").read_text(encoding="utf-8") assert leak.findall(control), "the pattern cannot find; the zeros below would mean nothing" tracked = [ SKILL, SKILL.parent / "references" / "README.md", SKILL.parent / "references" / "example-payload.json", PROJECT_ROOT / "tools" / "okf_consume.py", PROJECT_ROOT / "tools" / "okf_consume_measure.py", PROJECT_ROOT / "tests" / "test_okf_consume.py", PROJECT_ROOT / "docs" / "2026-09-07-okf-konsumskill-maaling.md", PROJECT_ROOT / "docs" / "2026-09-08-blindsone-below-k-k2.md", PROJECT_ROOT / "docs" / "2026-09-08-blindsone-laas2-budsjett-k2.md", PROJECT_ROOT / "docs" / "2026-09-08-prisform-og-loggen-k2.md", PROJECT_ROOT / "docs" / "2026-09-08-kravnummer-tokenisering.md", PROJECT_ROOT / "docs" / "2026-09-08-sjeldenhetsvekt.md", PROJECT_ROOT / "docs" / "2026-09-08-claude-code-skill-vilkaarlig-bundle.md", PROJECT_ROOT / "README.md", PROJECT_ROOT / "CLAUDE.md", ] for path in tracked: assert leak.findall(path.read_text(encoding="utf-8")) == [], path def _quota_concept(concept_id: str, *, source_file: str) -> okf_consume.Concept: """A minimal concept whose only interesting property is its source document.""" return okf_consume.Concept( path=Path(concept_id), concept_id=concept_id, bundle_id="quota-fixture", bundle_id_inherited=False, sha256="0" * 64, okf_type="Krav", title=concept_id, source_file=source_file, adjudication="unknown", adjudication_present=False, req_number="", sources=(), sources_present=False, locators={}, frontmatter={}, body="alpha beta gamma", ) def test_a_source_quota_caps_how_many_places_one_document_takes() -> None: """One source document taking most of the payload is a MEASURED defect. Measured outside this repo on a 3206-concept bundle of a published handbook: the code's own process overview contributes 28 of 3206 concepts (0.87 %) and 8.0 % of the source characters, and takes 22 of 43 delivered places on one question and 24 of 45 on the known-positive. Identical at 343 and 1651 concepts, so it is the corpus's COMPOSITION -- that it holds its own table of contents -- and not its size. The quota cuts where the shortlist is cut, so `k` is still delivered in full: a concept over quota is withheld by NAME and the place goes to the next candidate. """ concepts = tuple( _quota_concept(f"c{index}", source_file="o.md" if index < 6 else f"x{index}.md") for index in range(10) ) ranked = tuple((concept, 1.0 / (index + 1), 3) for index, concept in enumerate(concepts)) delivered, withheld, _ = okf_consume.cut( ranked, k=4, limit=okf_consume.DEFAULT_LIMIT, source_quota=2 ) ids = [str(entry["concept_id"]) for entry in delivered] assert [cid for cid in ids if cid in {"c0", "c1", "c2", "c3", "c4", "c5"}] == ["c0", "c1"] assert len(delivered) == 4, "the freed places are filled, never left empty" assert ("c2", "source_quota_exceeded") in withheld assert len(withheld) + len(delivered) == len(concepts) def test_the_quota_never_shortens_a_single_document_bundle_s_payload() -> None: """The adverse case, recorded rather than discovered by a consumer. Every concept in a one-document bundle shares a `source_file`, so a quota applied literally would deliver `source_quota` excerpts where `k` were asked for. The top-up takes the best-ranked over-quota candidates back, so such a bundle is byte-identical to the quota being off. """ concepts = tuple(_quota_concept(f"c{index}", source_file="only.md") for index in range(10)) ranked = tuple((concept, 1.0 / (index + 1), 3) for index, concept in enumerate(concepts)) with_quota, _, _ = okf_consume.cut(ranked, k=4, limit=okf_consume.DEFAULT_LIMIT, source_quota=2) without, _, _ = okf_consume.cut(ranked, k=4, limit=okf_consume.DEFAULT_LIMIT) assert len(with_quota) == 4 assert with_quota == without def test_the_quota_has_an_explicit_opt_out_that_restores_the_old_order() -> None: """`None` is the opt-out, and it must reproduce the pre-round-11 cut. The default is 2, so this is the direction that needs proving: a consumer pinned to the previous excerpt order passes `--no-source-quota` and gets exactly what it got before, including no entry under the new rule. """ concepts = tuple( _quota_concept(f"c{index}", source_file="o.md" if index < 6 else f"x{index}.md") for index in range(10) ) ranked = tuple((concept, 1.0 / (index + 1), 3) for index, concept in enumerate(concepts)) delivered, withheld, _ = okf_consume.cut( ranked, k=4, limit=okf_consume.DEFAULT_LIMIT, source_quota=None ) assert [str(entry["concept_id"]) for entry in delivered] == ["c0", "c1", "c2", "c3"] assert all(rule != "source_quota_exceeded" for _, rule in withheld) def test_the_pre_pass_signature_defaults_are_the_consume_command_defaults() -> None: """One flag, one default -- the O6 defect class, on the reading side. `build_payload` is called as a Python function by the pin tests, the skill generator and any consumer that imports this package; `okf consume` is called by everyone else. When a flag's argparse default and its signature default disagree, those two populations get different payloads from the same version, and every measurement report is pinned to whichever one the measuring script happened to use. `okf project` shipped that defect for two rounds before O6 measured it. """ parsed = okf_consume.parse_args(["bundle", "--question", "q"]) signature = inspect.signature(okf_consume.build_payload) disagreeing = { name: (parameter.default, getattr(parsed, name)) for name, parameter in signature.parameters.items() if hasattr(parsed, name) and parameter.default is not inspect.Parameter.empty and getattr(parsed, name) != parameter.default } assert disagreeing == {} def test_the_quota_default_is_two_and_the_default_is_a_measurement() -> None: """The published default, held by a test that goes red if it drifts. Swept over {2, 3, 4, off} on three bundles: 2 and 3 take hit@8 from 5 of 6 to 6 of 6 on both K2 bundles with all five standing rank-1 rows unmoved, 4 and off leave it at 5 of 6. 2 rather than 3 on rank -- the recovered rows come in at 5 and 4 rather than 7 and 5. """ assert okf_consume.DEFAULT_SOURCE_QUOTA == 2 def test_the_readme_consume_section_states_the_rule_count_the_code_emits() -> None: # A published number must have a test that goes red when it goes false. readme = (PROJECT_ROOT / "README.md").read_text(encoding="utf-8") assert readme.count("## Consume\n") == 1 assert readme.count("## Consume in Claude Code\n") == 1 assert len(okf_consume.WITHHOLDING_RULES) == 7 assert "closed set of seven" in readme assert "tools/okf_consume.py" in readme def test_the_readme_recipe_names_only_commands_this_repository_ships() -> None: """Every command the recipe invokes must exist. What "exist" means MOVED. Until 2026-09-08 (O5) the recipe told a reader to run `python3 tools/