llm-ingestion-okf/tests/test_segmented_collisions.py
Kjell Tore Guttormsen 9d1f4b14ed test(fixtures): replace sector-specific example material with generic, fictitious examples — green
Every fixture, test document, tool example and document now uses an invented
kitchen-and-baking handbook series, written in this repository. The package's
behaviour is unchanged; src/ changes are comments and help text only.

- Generated fixtures are regenerated from their generators. Their structural
  counts are identical before and after: elements, images, rows, cells,
  headings, bookmarks and the witness inventory's per-document totals. The
  image-inbox and accounting documents are renamed kapittel-84-*.
- tools/okf_accounting_gate.py: the two options that named one real corpus
  each are replaced by a generic, repeatable --corpus PATH with no default.
  Row 5 compares the PDF pair alone. Gate verdict unchanged: RED rows 2, 3, 6.
- tools/okf_witness.py: the STS JSON reader for one publisher's delivery is
  removed, along with its three twins and five tests. The mutation harness
  loses W09.
- docs/: 13 dated reports that documented runs on a retired reference corpus
  are removed, and 40 are neutralized. Dead links are removed, and no new
  dangling path is introduced.
- The synthetic MCP-gate corpus and the residual probe words are neutral.

Valgt: keep the `okf quality --fasit` bar value (the measured fraction, one corpus) and
rewrite only its provenance, because the verdict stays unchanged and the
number names nothing.

Term check with the local list: 0 of 411 tracked files, 0 file names, 0 of
27 binary fixtures. Suite after git add: 2457 passed, 1 skipped. The base
tree had 2460 passed and 2 skipped; five tests went with the JSON reader and
four were added by the term check. ruff, ruff format and mypy --strict src/
are clean.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-23 14:52:02 +02:00

215 lines
8.2 KiB
Python

"""The collision gate under 1-to-N, and reporting that stays additive.
Today's gate names every dropped file BEFORE any gate call or write, so two
files reducing to one slug are refused TOGETHER rather than letting iteration
order decide which one survives. Under segmentation a document no longer claims
one name -- it claims the whole SET of paths its plan expands to -- so the gate
has to be keyed on that set or the same defect returns one level down: the
second document silently claims the first's concepts.
Two length rules that look alike and are not. `check_filename_length` measures
ONE name against NAME_MAX, which is a per-directory-entry limit. Measuring a
joined hierarchical path against it gets the question backwards in both
directions: it would refuse a perfectly legal deep path, and it would accept an
illegal component sitting in a short one.
Reporting is extended ADDITIVELY. `persisted` keeps its per-source-file
meaning, so an existing consumer reading it sees exactly what it saw before,
and the per-concept expansion arrives as a new `concepts` field beside it.
"""
from __future__ import annotations
import hashlib
from pathlib import Path
from typing import Any
from llm_ingestion_okf.extract import extract_text
from llm_ingestion_okf.inbox import GateDecision, process_inbox
from llm_ingestion_okf.materialize import NAME_MAX_BYTES
from llm_ingestion_okf.profiles import DEFAULT, SEGMENTED_V1
from llm_ingestion_okf.segmentation import (
SegmentationPlan,
observed_extractor_version,
parse_segmentation_plan,
)
INGESTED_AT = "2026-07-25T12:00:00Z"
PLAN_AT = "2026-08-30T09:00:00Z"
# Long enough that every span this module declares (`index * 10` onward)
# lands inside it -- a span past the end is a different refusal, and would
# mask the collision these tests are about.
DOCUMENT = "Krav i konseptet.\n" * 20
def gate(text: str) -> GateDecision:
return GateDecision(sanitized_text=text, disposition="warn")
def drop(inbox: Path, name: str, text: str = DOCUMENT) -> Path:
inbox.mkdir(parents=True, exist_ok=True)
path = inbox / name
path.write_text(text, encoding="utf-8", newline="")
return path
def _extracted_text_sha256(source_bytes: bytes, filename: str = "q500.md") -> str:
return hashlib.sha256(extract_text(filename, source_bytes).encode("utf-8")).hexdigest()
def build_plan(source_bytes: bytes, paths: tuple[str, ...], **overrides: Any) -> SegmentationPlan:
payload: dict[str, Any] = {
"version": "1",
"source_sha256": hashlib.sha256(source_bytes).hexdigest(),
# The hash of the CANONICAL EXTRACTED text, which is what the spans
# index. Equal to the source hash on a `.md` passthrough and computed
# rather than copied, so the fixture keeps saying which one it means.
"text_sha256": _extracted_text_sha256(source_bytes),
"extractor_id": "md",
"extractor_version": observed_extractor_version("md"),
"adjudicated_at": "2026-08-30T08:00:00Z",
"entries": [
{
"segment_id": f"s{index}",
"path": path,
"title": f"Del {index}",
"okf_type": "requirement",
"span": [index * 10, index * 10 + 10],
"ingested_at": PLAN_AT,
}
for index, path in enumerate(paths)
],
}
payload.update(overrides)
return parse_segmentation_plan(payload)
def run(
tmp: Path,
*,
plan: SegmentationPlan | None = None,
profile=SEGMENTED_V1,
round_name: str = "round",
):
return process_inbox(
tmp / round_name,
tmp / "bundle",
INGESTED_AT,
okf_type="requirement",
gate=gate,
profile=profile,
root_frontmatter_values={"bundle_id": "b-1"} if profile is SEGMENTED_V1 else None,
segmentation=plan,
)
def tree(bundle: Path) -> dict[str, bytes]:
if not bundle.is_dir():
return {}
return {
str(path.relative_to(bundle)): path.read_bytes()
for path in sorted(bundle.rglob("*"))
if path.is_file()
}
# --- the gate, keyed on the whole set of segment paths --------------------
def test_two_documents_claiming_one_segment_path_are_both_refused(tmp_path: Path) -> None:
# Identical bytes, so ONE plan covers both documents and both expand onto
# the same paths. Refused together, before any gate call or write.
drop(tmp_path / "round", "q500.md")
drop(tmp_path / "round", "w720.md")
plan = build_plan(
DOCUMENT.encode("utf-8"), ("krav/3-1/brannkonsept.md", "krav/3-2/roemning.md")
)
result = run(tmp_path, plan=plan)
assert {entry.error.code for entry in result.failed} == {"inbox_slug_collision"}
assert {entry.source_file for entry in result.failed} == {"q500.md", "w720.md"}
assert tree(tmp_path / "bundle") == {}
assert result.persisted == ()
def test_the_collision_message_names_the_contested_path(tmp_path: Path) -> None:
drop(tmp_path / "round", "q500.md")
drop(tmp_path / "round", "w720.md")
plan = build_plan(DOCUMENT.encode("utf-8"), ("krav/3-1/brannkonsept.md",))
result = run(tmp_path, plan=plan)
assert all("krav/3-1/brannkonsept.md" in str(entry.error) for entry in result.failed)
def test_a_flat_run_still_refuses_two_names_reducing_to_one_slug(tmp_path: Path) -> None:
drop(tmp_path / "round", "note.md", "a\n")
drop(tmp_path / "round", "note.txt", "b\n")
result = run(tmp_path, profile=DEFAULT)
assert {entry.error.code for entry in result.failed} == {"inbox_slug_collision"}
assert tree(tmp_path / "bundle") == {}
# --- the length rule is PER COMPONENT -------------------------------------
def test_a_single_over_long_component_is_refused(tmp_path: Path) -> None:
source = drop(tmp_path / "round", "q500.md")
too_long = "a" * (NAME_MAX_BYTES + 1)
plan = build_plan(source.read_bytes(), (f"krav/{too_long}.md",))
result = run(tmp_path, plan=plan)
assert {entry.error.code for entry in result.failed} == {"inbox_slug_too_long"}
assert tree(tmp_path / "bundle") == {}
def test_a_joined_path_over_name_max_with_legal_components_is_accepted(tmp_path: Path) -> None:
# The check the joined measurement gets backwards. Every component here is
# well under NAME_MAX; the joined path is well over it, and the filesystem
# does not care -- NAME_MAX is a per-entry limit.
source = drop(tmp_path / "round", "q500.md")
deep = "/".join(f"niva-{index}-{'x' * 40}" for index in range(6))
target = f"{deep}/krav.md"
assert len(target.encode("utf-8")) > NAME_MAX_BYTES
assert all(len(part.encode("utf-8")) <= NAME_MAX_BYTES for part in target.split("/"))
result = run(tmp_path, plan=build_plan(source.read_bytes(), (target,)))
assert result.failed == ()
assert target in tree(tmp_path / "bundle")
# --- additive reporting ---------------------------------------------------
def test_concepts_holds_one_entry_per_concept_and_persisted_one_per_source(
tmp_path: Path,
) -> None:
source = drop(tmp_path / "round", "q500.md")
plan = build_plan(source.read_bytes(), ("krav/a.md", "krav/b.md", "krav/c.md"))
result = run(tmp_path, plan=plan)
assert len(result.concepts) == 3
assert {entry.path.name for entry in result.concepts} == {"a.md", "b.md", "c.md"}
# `persisted` keeps its existing meaning: one entry per SOURCE FILE. A
# consumer reading it sees exactly what it saw before segmentation existed.
assert len(result.persisted) == 1
assert result.persisted[0].source_file == "q500.md"
def test_under_default_the_two_fields_agree(tmp_path: Path) -> None:
drop(tmp_path / "round", "q500.md")
drop(tmp_path / "round", "w720.md", "annet\n")
result = run(tmp_path, profile=DEFAULT)
assert len(result.persisted) == 2
assert result.concepts == result.persisted
def test_concepts_names_every_segment_path_it_wrote(tmp_path: Path) -> None:
source = drop(tmp_path / "round", "q500.md")
plan = build_plan(source.read_bytes(), ("krav/3-1/a.md", "krav/3-2/b.md"))
result = run(tmp_path, plan=plan)
bundle = tmp_path / "bundle"
assert {str(entry.path.relative_to(bundle)) for entry in result.concepts} == {
"krav/3-1/a.md",
"krav/3-2/b.md",
}
assert all(entry.source_file == "q500.md" for entry in result.concepts)