"""No character stands BETWEEN two concepts, or after the last one. Round 7 closed the text ABOVE the first concept (`--first-span-from-zero`, 163 804 characters, 79 % of the whole gap) and named the remainder without opening it: 43 631 characters, 2.51 % of the corpus, over 8 of the 32 documents that get a plan -- 26 041 between entries and 17 590 after the last. Measured here, the remainder has ONE cause, not the two the shape suggests. Every rule in `find_candidates` closes a span against the NEXT MARK, and three separate steps then remove a mark after its neighbour's `end` was already fixed against it: * the **orphan check** drops a heading with no body, and the heading line itself -- `## 3 Prising\n`, 14 characters -- is what the next candidate no longer starts at. Measured with the fold off: **18 527 characters over 15 of 39 documents**; * `fold_units` **clause 1** discards a contents run, and the text those entries opened goes with them. It accounts for the remaining **7 514** characters between entries; * the same clause discarding the LAST run takes the tail with it. Measured: with `unit_fold=False` the tail gap over the corpus is **0**, so all **17 590** tail characters are clause 1's. That is round 6's own principle -- the outline gate filters at ADMISSION, "before spans close, so the text a removed mark opened is carried by the mark above" -- not applied to the two steps that remove marks AFTER spans close. This rule states it once, at the end, where every removal has already happened: a candidate's span runs to the next candidate's start, and the last one runs to the end of the text. It adds NO boundary and removes none. A plan's entry count is identical either way, which is the same property `--first-span-from-zero` has and the reason both can be measured by coverage rather than by count. """ from __future__ import annotations from pathlib import Path from llm_ingestion_okf import cli from llm_ingestion_okf.propose import find_candidates ARMS = dict(outline_run=3, table_grid=True, unit_fold=True, first_span_from_zero=True) #: The orphan check's gap: `## Tom` has no body, so it is dropped -- and the #: seven characters of its own heading line belong to no segment. ORPHAN = "## Forste\nInnhold under.\n## Tom\n## Andre\nInnhold under.\n" #: Clause 1's gap: three page-numbered siblings are read as a contents list #: and discarded, and the body under them goes too. 130 of 159 characters. CONTENTS_TAIL = ( "## Innledning\nBrodtekst her.\n" "## Kapittel en 3\nNoe innhold.\n" "## Kapittel to 5\nMer innhold.\n" "## Kapittel tre 9\nEn avsluttende brodtekst som ingen kandidat dekker.\n" ) #: The known-negative: nothing is removed, so nothing is carried. NO_GAP = "## Forste\nInnhold under.\n## Andre\nInnhold under.\n" def _uncovered(text: str, **kwargs: object) -> int: candidates = find_candidates(text, **ARMS, **kwargs) # type: ignore[arg-type] if not candidates: return len(text) between = sum( max(0, candidates[i].start - candidates[i - 1].end) for i in range(1, len(candidates)) ) return candidates[0].start + between + max(0, len(text) - candidates[-1].end) def test_the_default_still_loses_the_orphaned_heading_line() -> None: """The defect, kept visible: the dropped mark's own line goes nowhere.""" assert _uncovered(ORPHAN) == len("## Tom\n") def test_the_default_still_loses_the_tail_under_a_discarded_contents_run() -> None: """82 % of this document is in no segment, and none of it is a listing.""" assert _uncovered(CONTENTS_TAIL) == 130 def test_no_character_lies_between_two_concepts() -> None: assert _uncovered(ORPHAN, close_span_gaps=True) == 0 candidates = find_candidates(ORPHAN, **ARMS, close_span_gaps=True) carried = ORPHAN[candidates[0].start : candidates[0].end] assert "## Tom" in carried, "the removed mark's text is carried by the mark ABOVE" def test_no_character_lies_after_the_last_concept() -> None: assert _uncovered(CONTENTS_TAIL, close_span_gaps=True) == 0 candidates = find_candidates(CONTENTS_TAIL, **ARMS, close_span_gaps=True) assert CONTENTS_TAIL[candidates[-1].start : candidates[-1].end].endswith( "En avsluttende brodtekst som ingen kandidat dekker.\n" ) def test_the_rule_adds_and_removes_no_boundary() -> None: """Only `end` moves. Titles, rules and starts are identical.""" for text in (ORPHAN, CONTENTS_TAIL, NO_GAP): plain = find_candidates(text, **ARMS) closed = find_candidates(text, **ARMS, close_span_gaps=True) assert [c.title for c in plain] == [c.title for c in closed] assert [c.rule for c in plain] == [c.rule for c in closed] assert [c.start for c in plain] == [c.start for c in closed] def test_a_document_with_no_gap_is_untouched() -> None: """The known-negative, as IDENTICAL objects rather than an equal count.""" before = find_candidates(NO_GAP, **ARMS) after = find_candidates(NO_GAP, **ARMS, close_span_gaps=True) assert before == after assert _uncovered(NO_GAP) == 0 def test_it_is_on_by_default_in_the_build_command() -> None: assert cli.DEFAULT_CLOSE_SPAN_GAPS is True def test_the_parser_carries_the_default_and_an_opt_out() -> None: base = ["build", "src", "--bundle", "out", "--bundle-id", "x"] assert cli.parse_args(base).close_span_gaps is True assert cli.parse_args([*base, "--no-close-span-gaps"]).close_span_gaps is False def test_the_flag_reaches_the_proposer_from_the_build_command(tmp_path: Path) -> None: """The plumbing, and it needs its own test rather than the smoke build. Measured 2026-09-11: the operator's five-document folder has a coverage gap of ZERO under round 7's default already, so building it with the rule on and off gives `diff -rq` 0 differences. That is the rule behaving correctly on an input with nothing to carry -- and it means that build proves nothing about whether the flag arrives. This one uses a document that HAS a gap. """ inbox = tmp_path / "in" inbox.mkdir() (inbox / "d.md").write_text(CONTENTS_TAIL, encoding="utf-8") def concepts(bundle: Path, **kwargs: object) -> str: report = cli.build( inbox, bundle=bundle, bundle_id="t", okf_version="0.2", ingested_at="1970-01-01T00:00:00Z", **kwargs, # type: ignore[arg-type] ) assert report.codes == () return "".join(path.read_text(encoding="utf-8") for path in sorted(bundle.rglob("*.md"))) on = concepts(tmp_path / "on") off = concepts(tmp_path / "off", close_span_gaps=False) assert "En avsluttende brodtekst som ingen kandidat dekker." in on assert "En avsluttende brodtekst som ingen kandidat dekker." not in off