"""The bundle the DEFAULT build produces, pinned where a regression goes red. `tests/test_okf_consume.py` pinned hit@8 against the Arm B bundle alone -- the configuration `okf build` stopped emitting on 2026-09-08. A published number measured on a bundle nobody produces is a number that cannot regress, so the guarantee it looks like was never held by anything. This file pins the CURRENT default: `--outline-run 3 --table-grid --unit-fold --drop-wrapped-outline --outline-gate --first-span-from-zero --sheet-section-rows --keep-table-heading --close-span-gaps`, plus the reading side's `tie_shared_rank`. Round 6 moved the first five on 2026-09-09, round 7 moved four more on 2026-09-10 and round 8 moved the last on 2026-09-11, each after measuring hit@8 on exactly the bundle its own default produces. The gold set is LOCAL-ONLY and stays that way: no question and no `gold_document` is reproduced here, and a row is named by its INDEX, the way `docs/2026-09-07-okf-konsumskill-maaling.md` already names them. The bundle itself is a build artefact, not a fixture: it is 832 files of a consumer's corpus and this repository is public. Absent, these tests SKIP with the command that rebuilds it -- "not measured", never zero. """ from __future__ import annotations import json import sys from pathlib import Path import pytest PROJECT_ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(PROJECT_ROOT / "tools")) import okf_consume # noqa: E402 import okf_consume_measure # noqa: E402 #: Built by: #: okf build /K2/trinn1 \ #: --bundle ~/corpora/okf-telling-20260829/K2-bundle-default-20260911 \ #: --bundle-id k2-trinn1-20260903 --okf-version 0.2 #: with no arm flag at all -- the package default, which is the point. #: #: Rebuilt 2026-09-09 for `--contents-name` (round 9). Digest, from inside the #: bundle: #: find . -type f -print0 | sort -z | xargs -0 shasum -a 256 | shasum -a 256 #: -> 21af4a1aa98315cf514c4cbc6b4a9b77ce63960224d6d7b31b34d55cc67fb2ad #: (The previous default, `K2-bundle-default-20260911`, was #: 8c93e5e3222577a2b3352ca83af980e403d3a571c3a467b83c3d8170b1df2b69 at 436 #: concepts and stays on disk.) #: Two independent builds of it differ in NOTHING (`diff -rq`), including #: `log.md`, which carries the corpus path and never the bundle's own. #: #: CONCEPT IDS MOVED IN THIS REBUILD, and not only because the count did. #: Round 9 strips pandoc's `{#sheet-N}` / `{#slide-N}` anchor where a title is #: formed, and a concept's filename is reduced FROM its title, so TWO ids on #: this bundle are renamed: #: del-ii-bilag-7-prisskjema/prissammenstilling-sheet-1 -> .../prissammenstilling #: del-ii-bilag-0-dokumentliste-del-ii/ark1-sheet-1 -> .../ark1 #: The first is an id `portfolio-optimiser` has cited in writing. The rename #: was authorised by the operator on 2026-09-09 after the exposure was counted: #: 2 of 810 concepts on the previous default and 2 of 1108 on Arm B. DEFAULT_BUNDLE = Path.home() / "corpora" / "okf-telling-20260829" / "K2-bundle-default-20260912" GOLD_SET = PROJECT_ROOT / ".claude/projects/2026-09-07-okf-consume-prepass/hit-at-k-questions.json" requires_default_bundle = pytest.mark.skipif( not DEFAULT_BUNDLE.is_dir() or not GOLD_SET.is_file(), reason=( f"the default-configuration K2 bundle is not present at {DEFAULT_BUNDLE}. " "NOT MEASURED, not zero: rebuild it with `okf build /K2/trinn1 " "--bundle --bundle-id k2-trinn1-20260903 --okf-version 0.2`" ), ) #: Measured 2026-09-09 on the bundle above. The count moved 425 -> 436 with #: `--sheet-section-rows --keep-table-heading`; `--first-span-from-zero` and #: `--close-span-gaps` each moved it by NOTHING, which is the point of both -- #: they add no boundary, they only move a span's start or its end. Round 8's #: rule closed 43 631 characters (2.51 % of the corpus) that were in no #: segment, and the count was byte-for-byte the same 436. #: #: 436 -> 453 with round 9's `--contents-name`, which does add concepts: a run #: of data rows is no longer read as a contents listing and discarded, so the #: candidates it was taking with it survive. Corpus-wide, 429 -> 447 candidates #: over 32 -> 33 documents with a plan, and characters in no segment stay 0. EXPECTED_CONCEPTS = 453 EXPECTED_HITS = 5 #: Rank per question INDEX, `None` for the row that misses on every bundle and #: every configuration measured so far. The identity is the index; the question #: stays in the local-only gold set. EXPECTED_RANKS = (1, 1, 1, 1, 1, None) @requires_default_bundle def test_the_default_bundle_holds_its_concept_count() -> None: assert len(list(okf_consume.enumerate_concepts(DEFAULT_BUNDLE))) == EXPECTED_CONCEPTS @requires_default_bundle def test_hit_at_eight_holds_rank_one_on_every_row_it_held() -> None: """The acceptance criterion round 6's default move had to clear. Not the hit COUNT alone: the count survived a configuration that lost a row from rank 1 to rank 2, which is exactly how the previous round's regression hid. The rank per row is the pin. On THIS bundle that is not a hypothetical -- see the test below, which reproduces the fall on these exact bytes by turning the reading-side default off. """ questions = json.loads(GOLD_SET.read_text(encoding="utf-8"))["questions"] assert len(questions) == len(EXPECTED_RANKS), "the gold set changed shape" ranks = [] for entry in questions: payload = okf_consume.build_payload(DEFAULT_BUNDLE, question=entry["question"]) excerpts = payload["excerpts"] assert isinstance(excerpts, list) ranks.append(okf_consume_measure.hit_rank(excerpts, entry["gold_document"])) assert tuple(ranks) == EXPECTED_RANKS, f"hit@8 ranks moved: {ranks}" assert sum(rank is not None for rank in ranks) == EXPECTED_HITS @requires_default_bundle def test_the_bundle_declares_the_identity_the_reader_needs() -> None: """Whatever else moves, the bundle stays one the reading direction opens.""" assert okf_consume.root_bundle_id_of(DEFAULT_BUNDLE) == "k2-trinn1-20260903" @requires_default_bundle def test_the_reading_default_is_what_holds_row_one_on_these_bytes() -> None: """The known-negative, on the shipped bundle rather than a fixture. Round 7 moved `--sheet-section-rows --keep-table-heading` into the build default, which splits row 1's gold document from 1 concept into 12. Round 6 measured that exact split costing row 1 its rank, and held the two rules back for it. What removed the cost is `consume.DEFAULT_TIE_SHARED_RANK`, and this test is the proof that it is still what removes it: turn it off on these bytes and the fall comes back. Without this, `EXPECTED_RANKS` above would be a green assertion with no stated cause, and a later change to the fusion could take the cause away while the pin stayed green on some other accident. """ questions = json.loads(GOLD_SET.read_text(encoding="utf-8"))["questions"] ranks = [] for entry in questions: payload = okf_consume.build_payload( DEFAULT_BUNDLE, question=entry["question"], tie_shared_rank=False ) excerpts = payload["excerpts"] assert isinstance(excerpts, list) ranks.append(okf_consume_measure.hit_rank(excerpts, entry["gold_document"])) assert ranks[0] == 2, "the known-negative stopped being negative" assert tuple(ranks[1:]) == EXPECTED_RANKS[1:]