"""A structure the publisher DECLARED takes the route declared structure takes. MEASURED, AND THE MEASUREMENT IS THE WHOLE REASON THIS FILE EXISTS. Round 13 gave `.xml` a reader that emits **2 761 of 2 761** ATX lines for R761's titled `` elements, with the text preserved exactly. At SHIPPED DEFAULTS the build then delivered **23 concepts and 15 of 2 761 boundaries** -- everything after the reader ate it. Two steps, each measured on that document and reproduced in miniature by `sts-container.xml`: - the **orphan check** removes a heading with nothing under its own first line. It took **710 of 2 761**, and 710 of 710 removed headings are followed immediately by another heading while **0 of 2 051** delivered ones are: they are container sections, not false positives. The check asks whether a GUESS was a heading, which is the wrong question for a section a publisher named. - **Arm F** (`--unit-fold`, a build default) folds every level deeper than the shallowest repeated one into its parent. It took **2 066 more**, 2 089 -> 23, because a process code whose chapters are all `14.121 ...` reads as a unit sheet. `find_candidates` already skips both for `outline_marks`, the PDF bookmark arm -- which is why that arm reaches 2 762. This file pins the same route for the one type whose reader KNOWS the structure because it wrote the heading itself, under its OWN rule name, so an artifact still says which of the two a boundary came from. Nothing else may move: the rule is reached only from the `.xml` row. `sts-container.xml` is hand-written and carries no sentence from any source. It reproduces both mechanisms at 9 declared sections: Arm B delivers 8 (the orphan check takes the container) and the build defaults deliver 3. """ from __future__ import annotations from pathlib import Path from llm_ingestion_okf import extract, propose FIXTURES = Path(__file__).parent / "fixtures" STS = FIXTURES / "sts-mini.xml" CONTAINER = FIXTURES / "sts-container.xml" GENERIC = FIXTURES / "generic-feed.xml" #: Every flag `okf build` turns on by default, so a test that says "at shipped #: defaults" is measuring the defaults and not a subset of them. BUILD_DEFAULTS = dict( outline_run=3, table_grid=True, unit_fold=True, keep_table_heading=True, sheet_section_rows=True, drop_wrapped_outline=True, outline_gate=True, first_span_from_zero=True, close_span_gaps=True, contents_name=True, ) def test_the_reader_reports_the_line_every_declared_section_landed_on() -> None: """No bridge, and that is the difference from the PDF arm. The PDF arm has to map (page, `/XYZ` top) onto a line index and was wrong on 1 840 of 2 762 under the naive rule. Here the reader WROTE the line, so the index is the one it appended at -- asserted against the text itself rather than against a count, because a mark naming the right title at the wrong offset passes any count. """ text = extract.extract_text(CONTAINER.name, CONTAINER.read_bytes()) marks = extract.xml_outline(CONTAINER.name, CONTAINER.read_bytes()) lines = text.split("\n") assert [(m.level, m.title) for m in marks] == [ (1, "1 Kapittel uten egen brodtekst"), (2, "1.1 Foerste underseksjon"), (2, "1.2 Andre underseksjon"), (1, "2 Kapittel med brodtekst"), (2, "2.1 Enhet oppgjort etter lengde"), (2, "2.2 Enhet oppgjort etter areal"), (2, "2.3 Enhet oppgjort etter volum"), (1, "3 Tredje kapittel"), (2, "3.1 Enhet oppgjort etter stykk"), ] for mark in marks: assert lines[mark.line] == "#" * mark.level + " " + mark.title def test_a_declared_section_is_a_rule_of_its_own_and_not_the_bookmark_arms() -> None: """A reader who cannot tell them apart in the artifact cannot tell them apart.""" assert propose.RULE_XML_SECTION == "rule:xml-section" assert propose.RULE_XML_SECTION in propose.RULE_NAMES assert propose.RULE_XML_SECTION != propose.RULE_PDF_OUTLINE assert propose.RULE_XML_SECTION != propose.RULE_HEADING def test_a_container_section_survives_where_the_orphan_check_deletes_it() -> None: """710 of 2 761 on R761; 1 of 9 here, and it is the chapter above the rest.""" text = extract.extract_text(CONTAINER.name, CONTAINER.read_bytes()) marks = extract.xml_outline(CONTAINER.name, CONTAINER.read_bytes()) without = [c.title for c in propose.find_candidates(text)] assert "1 Kapittel uten egen brodtekst" not in without declared = propose.find_candidates( text, outline_marks=marks, outline_rule=propose.RULE_XML_SECTION ) assert [c.title for c in declared] == [ "Kapittel uten egen brodtekst", "Foerste underseksjon", "Andre underseksjon", "Kapittel med brodtekst", "Enhet oppgjort etter lengde", "Enhet oppgjort etter areal", "Enhet oppgjort etter volum", "Tredje kapittel", "Enhet oppgjort etter stykk", ] assert {c.rule for c in declared} == {propose.RULE_XML_SECTION} def test_the_unit_fold_does_not_eat_a_section_the_publisher_declared() -> None: """2 089 -> 23 on R761; 8 -> 3 here. At the SHIPPED defaults, not behind a flag.""" text = extract.extract_text(CONTAINER.name, CONTAINER.read_bytes()) marks = extract.xml_outline(CONTAINER.name, CONTAINER.read_bytes()) assert len(propose.find_candidates(text, **BUILD_DEFAULTS)) == 3 assert ( len( propose.find_candidates( text, outline_marks=marks, outline_rule=propose.RULE_XML_SECTION, **BUILD_DEFAULTS, ) ) == 9 ) def test_only_the_xml_row_takes_the_declared_route(tmp_path: Path) -> None: """The same markdown from a `.md` file is a GUESS and keeps `rule:heading`. The route is chosen by the row, not by the text, which is what keeps every other file type byte-identical: a document whose reader did not write the heading has declared nothing. """ text = extract.extract_text(CONTAINER.name, CONTAINER.read_bytes()) source = tmp_path / "same-markdown.md" source.write_text(text, encoding="utf-8") plan = propose.build_plan( source, text, source.read_bytes(), okf_type="reference", proposed_at="2026-01-01T00:00:00Z", **BUILD_DEFAULTS, ) assert [entry["derived"][1] for entry in plan["entries"]] == [propose.RULE_HEADING] * 3 xml_source = tmp_path / "sts-container.xml" xml_source.write_bytes(CONTAINER.read_bytes()) xml_plan = propose.build_plan( xml_source, text, xml_source.read_bytes(), okf_type="reference", proposed_at="2026-01-01T00:00:00Z", **BUILD_DEFAULTS, ) assert [entry["derived"][1] for entry in xml_plan["entries"]] == [propose.RULE_XML_SECTION] * 9 def test_the_number_becomes_the_directory_the_way_the_bookmark_arm_does_it() -> None: """`