609 lines
24 KiB
Python
609 lines
24 KiB
Python
"""One document becomes many concepts: Door B's 1-to-N path.
|
|
|
|
The measured defect this closes. `process_inbox` emitted exactly one flat
|
|
concept per dropped file, which is the shape OKF v0.2's Appendix A presents
|
|
v0.1 as migrating AWAY from -- and §11 could never catch it, because it checks
|
|
that every non-reserved `.md` parses with a non-empty `type`. A bundle of one
|
|
giant concept is fully conformant. Conformance is the floor, not the proof.
|
|
|
|
Two invariants are load-bearing here and are asserted rather than assumed:
|
|
|
|
- **Every segment is gated BEFORE any segment is written.** Gating and writing
|
|
one at a time would leave a half-screened document on disk the moment
|
|
segment 3 of 5 quarantines -- part of a document the guard refused, persisted
|
|
and indexed, with the run reporting success.
|
|
- **A span indexes the CANONICAL EXTRACTED TEXT.** The `.csv` fixture below is
|
|
what proves it: `render_table` re-renders those bytes, so a span computed
|
|
against the source bytes lands on different characters. On a `.md` fixture
|
|
bytes and text are identical and the assertion would pass for the wrong
|
|
reason.
|
|
|
|
`csv` is a CORE type, needing no `[extract]` extra, so that fixture can never
|
|
skip. A `pdf` would be the trap: pdfplumber is installed in this venv, so it
|
|
would pass here and skip silently in a bare consumer environment.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf.errors import SegmentationError
|
|
from llm_ingestion_okf.extract import extract_text
|
|
from llm_ingestion_okf.inbox import GateDecision, process_inbox
|
|
from llm_ingestion_okf.profiles import DEFAULT, SEGMENTED_V1
|
|
from llm_ingestion_okf.segmentation import (
|
|
STDLIB_EXTRACTOR_VERSION,
|
|
SegmentationPlan,
|
|
observed_extractor_version,
|
|
parse_segmentation_plan,
|
|
)
|
|
|
|
INGESTED_AT = "2026-07-25T12:00:00Z"
|
|
PLAN_AT = "2026-08-30T09:00:00Z"
|
|
|
|
DOCUMENT = (
|
|
"0 Forord: bakgrunn for konseptet.\n"
|
|
"1 Brannkonsept: krav til seksjonering.\n"
|
|
"2 Roemning: to uavhengige veier.\n"
|
|
"3 Baereevne: R60 for hovedbaeresystem.\n"
|
|
"4 Slokkeanlegg: sprinkler i hele bygget.\n"
|
|
)
|
|
|
|
PATHS = (
|
|
"krav/forord.md",
|
|
"krav/3-1/brannkonsept.md",
|
|
"krav/3-1/roemning.md",
|
|
"krav/3-2/baereevne.md",
|
|
"krav/3-2/slokkeanlegg.md",
|
|
)
|
|
|
|
|
|
def gate(text: str) -> GateDecision:
|
|
return GateDecision(sanitized_text=text, disposition="warn")
|
|
|
|
|
|
def drop(inbox: Path, name: str, text: str) -> Path:
|
|
inbox.mkdir(parents=True, exist_ok=True)
|
|
path = inbox / name
|
|
path.write_text(text, encoding="utf-8", newline="")
|
|
return path
|
|
|
|
|
|
def line_spans(text: str) -> tuple[tuple[int, int], ...]:
|
|
"""Half-open spans, one per line, measured rather than hard-coded."""
|
|
spans: list[tuple[int, int]] = []
|
|
offset = 0
|
|
for line in text.splitlines(keepends=True):
|
|
spans.append((offset, offset + len(line)))
|
|
offset += len(line)
|
|
return tuple(spans)
|
|
|
|
|
|
def build_plan(
|
|
source_bytes: bytes,
|
|
text: str,
|
|
*,
|
|
paths: tuple[str, ...] = PATHS,
|
|
extractor_id: str = "md",
|
|
**overrides: Any,
|
|
) -> SegmentationPlan:
|
|
spans = line_spans(text)
|
|
payload: dict[str, Any] = {
|
|
"version": "1",
|
|
"source_sha256": hashlib.sha256(source_bytes).hexdigest(),
|
|
"text_sha256": hashlib.sha256(text.encode("utf-8")).hexdigest(),
|
|
"extractor_id": extractor_id,
|
|
"extractor_version": observed_extractor_version(extractor_id),
|
|
"adjudicated_at": "2026-08-30T08:00:00Z",
|
|
"entries": [
|
|
{
|
|
"segment_id": f"s{index}",
|
|
"path": path,
|
|
"title": f"Del {index}",
|
|
"okf_type": "requirement",
|
|
"span": list(spans[index]),
|
|
"ingested_at": PLAN_AT,
|
|
}
|
|
for index, path in enumerate(paths)
|
|
],
|
|
}
|
|
payload.update(overrides)
|
|
return parse_segmentation_plan(payload)
|
|
|
|
|
|
def run(
|
|
tmp: Path,
|
|
*,
|
|
plan: SegmentationPlan | None,
|
|
plans: dict[str, SegmentationPlan] | None = None,
|
|
profile=SEGMENTED_V1,
|
|
values: dict[str, str] | None = None,
|
|
round_name: str = "round",
|
|
guard=gate,
|
|
):
|
|
return process_inbox(
|
|
tmp / round_name,
|
|
tmp / "bundle",
|
|
INGESTED_AT,
|
|
okf_type="requirement",
|
|
gate=guard,
|
|
profile=profile,
|
|
root_frontmatter_values={"bundle_id": "b-1"} if values is None else values,
|
|
segmentation=plan,
|
|
segmentations=plans,
|
|
)
|
|
|
|
|
|
def concepts(bundle: Path, profile=SEGMENTED_V1) -> dict[str, str]:
|
|
"""Every non-reserved concept, keyed by bundle-relative path.
|
|
|
|
Reserved names are excluded at EVERY level, not just the root -- a
|
|
per-directory index writer puts one in each directory, and counting those
|
|
as concepts would inflate the count the moment nesting appeared.
|
|
"""
|
|
if not bundle.is_dir():
|
|
return {}
|
|
return {
|
|
str(path.relative_to(bundle)): path.read_text(encoding="utf-8")
|
|
for path in sorted(bundle.rglob(f"*{profile.paths.concept_suffix}"))
|
|
if path.is_file() and path.name != profile.index.name
|
|
}
|
|
|
|
|
|
def tree(bundle: Path) -> dict[str, bytes]:
|
|
if not bundle.is_dir():
|
|
return {}
|
|
return {
|
|
str(path.relative_to(bundle)): path.read_bytes()
|
|
for path in sorted(bundle.rglob("*"))
|
|
if path.is_file()
|
|
}
|
|
|
|
|
|
def body_of(document: str) -> str:
|
|
return document.split("---\n", 2)[2].lstrip("\n")
|
|
|
|
|
|
def frontmatter_of(document: str) -> dict[str, str]:
|
|
head = document.split("---\n")[1]
|
|
return dict(
|
|
line.split(": ", 1) for line in head.splitlines() if ": " in line and line[:1] != " "
|
|
)
|
|
|
|
|
|
# --- S1: one document, exactly N concepts ---------------------------------
|
|
|
|
|
|
def test_a_five_entry_plan_yields_exactly_five_concepts(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT))
|
|
found = concepts(tmp_path / "bundle")
|
|
assert len(found) == 5
|
|
assert len(found) > 1
|
|
assert set(found) == set(PATHS)
|
|
|
|
|
|
def test_without_a_plan_the_same_fixture_yields_exactly_one_concept(tmp_path: Path) -> None:
|
|
# The discriminating negative control. Without it, a test that counts five
|
|
# concepts proves nothing about whether the PLAN caused the split.
|
|
drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
run(tmp_path, plan=None)
|
|
found = concepts(tmp_path / "bundle")
|
|
assert len(found) == 1
|
|
assert set(found) == {"inbox-n500.md"}
|
|
|
|
|
|
def test_each_segment_id_maps_to_the_path_its_entry_declares(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT)
|
|
run(tmp_path, plan=plan)
|
|
found = concepts(tmp_path / "bundle")
|
|
mapped = {frontmatter_of(document)["segment_id"]: path for path, document in found.items()}
|
|
assert mapped == {item.segment_id: item.path for item in plan.entries}
|
|
|
|
|
|
def test_each_body_equals_the_span_its_own_entry_declares(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT)
|
|
run(tmp_path, plan=plan)
|
|
found = concepts(tmp_path / "bundle")
|
|
for item in plan.entries:
|
|
start, end = item.span
|
|
assert body_of(found[item.path]) == DOCUMENT[start:end]
|
|
|
|
|
|
def test_concepts_land_on_nested_paths_across_several_directories(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT))
|
|
found = concepts(tmp_path / "bundle")
|
|
directories = {str(Path(path).parent) for path in found}
|
|
assert len(directories) >= 2
|
|
nested = [path for path in found if "/" in path]
|
|
assert nested
|
|
# The concept ID is the bundle-relative path minus the suffix -- OKF v0.2
|
|
# §2's definition, not a name we assign.
|
|
for path in nested:
|
|
assert path.endswith(".md")
|
|
assert (tmp_path / "bundle" / path).is_file()
|
|
|
|
|
|
def test_the_plan_timestamp_reaches_every_concept(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT))
|
|
for document in concepts(tmp_path / "bundle").values():
|
|
keys = frontmatter_of(document)
|
|
assert keys["ingested_at"] == PLAN_AT
|
|
assert keys["bundle_id"] == "b-1"
|
|
|
|
|
|
# --- S9: the span indexes EXTRACTED text, not source bytes ----------------
|
|
|
|
|
|
CSV_SOURCE = "krav;beskrivelse\n3-1;seksjonering\n3-2;roemning\n"
|
|
|
|
|
|
def test_a_csv_body_equals_the_span_of_the_extracted_text(tmp_path: Path) -> None:
|
|
# `render_table` re-renders these bytes, so extracted text != source bytes.
|
|
# A span computed against the bytes would land on different characters and
|
|
# produce a concept nobody adjudicated, with nothing failing.
|
|
source = drop(tmp_path / "round", "krav.csv", CSV_SOURCE)
|
|
text = extract_text("krav.csv", source.read_bytes())
|
|
assert text != CSV_SOURCE
|
|
|
|
plan = build_plan(
|
|
source.read_bytes(),
|
|
text,
|
|
paths=tuple(f"tabell/rad-{index}.md" for index in range(len(line_spans(text)))),
|
|
extractor_id="csv",
|
|
)
|
|
run(tmp_path, plan=plan)
|
|
found = concepts(tmp_path / "bundle")
|
|
assert len(found) == len(plan.entries)
|
|
for item in plan.entries:
|
|
start, end = item.span
|
|
assert body_of(found[item.path]) == text[start:end]
|
|
|
|
|
|
# --- fail-fast misuse, before any disk mutation ---------------------------
|
|
|
|
|
|
def test_a_plan_against_a_profile_without_the_capability_is_refused(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT), profile=DEFAULT, values={})
|
|
assert excinfo.value.code == "segmentation_unsupported_profile"
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_a_plan_without_a_bundle_id_is_refused_before_any_write(tmp_path: Path) -> None:
|
|
# The assertion Step 5 deferred to here: `process_inbox` gains its
|
|
# `segmentation` parameter at THIS step, so this is the first point at
|
|
# which the required-bundle_id branch can be reached at all.
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT), values={})
|
|
assert excinfo.value.code == "segmentation_plan_invalid"
|
|
assert "bundle_id" in str(excinfo.value)
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_a_refused_plan_leaves_an_existing_bundle_untouched(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "one", "n500.md", DOCUMENT)
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT), round_name="one")
|
|
before = tree(tmp_path / "bundle")
|
|
assert before != {}
|
|
|
|
drop(tmp_path / "two", "v720.md", DOCUMENT)
|
|
with pytest.raises(SegmentationError):
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT), values={}, round_name="two")
|
|
assert tree(tmp_path / "bundle") == before
|
|
|
|
|
|
# --- the whole-document refusal -------------------------------------------
|
|
|
|
|
|
def test_one_quarantined_segment_persists_nothing_for_that_document(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT)
|
|
third = DOCUMENT[plan.entries[2].span[0] : plan.entries[2].span[1]]
|
|
|
|
def quarantining(text: str) -> GateDecision:
|
|
if text == third:
|
|
return GateDecision(
|
|
sanitized_text=text, disposition="quarantine_review", reasons=("segment 3",)
|
|
)
|
|
return GateDecision(sanitized_text=text, disposition="warn")
|
|
|
|
result = run(tmp_path, plan=plan, guard=quarantining)
|
|
assert concepts(tmp_path / "bundle") == {}
|
|
assert len(result.quarantined) == 1
|
|
assert result.quarantined[0].source_file == "n500.md"
|
|
assert result.persisted == ()
|
|
|
|
|
|
def test_every_segment_is_gated_before_any_is_written(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT)
|
|
calls: list[str] = []
|
|
written_when_gated: list[int] = []
|
|
|
|
def recording(text: str) -> GateDecision:
|
|
calls.append(text)
|
|
written_when_gated.append(len(tree(tmp_path / "bundle")))
|
|
return GateDecision(sanitized_text=text, disposition="warn")
|
|
|
|
run(tmp_path, plan=plan, guard=recording)
|
|
assert len(calls) == 5
|
|
# Nothing on disk while any gate call is still outstanding.
|
|
assert written_when_gated == [0, 0, 0, 0, 0]
|
|
assert len(concepts(tmp_path / "bundle")) == 5
|
|
|
|
|
|
def test_a_plan_covering_one_of_two_documents_leaves_the_other_flat(tmp_path: Path) -> None:
|
|
# DIFFERENT bytes, deliberately. A plan is selected by content hash, so two
|
|
# files with identical content are both covered by one plan and land every
|
|
# segment on the same path -- which is the intra-run collision Step 8's gate
|
|
# exists to refuse, not something to demonstrate here.
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
drop(tmp_path / "round", "v720.md", "V720 Tunnel: egne krav.\n")
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT))
|
|
found = concepts(tmp_path / "bundle")
|
|
assert set(PATHS) <= set(found)
|
|
assert "inbox-v720.md" in found
|
|
assert len(found) == 6
|
|
|
|
|
|
# --- a plan that matches nothing is refused, never silent ------------------
|
|
|
|
|
|
def test_a_plan_matching_no_dropped_file_is_refused(tmp_path: Path) -> None:
|
|
"""The silent skip this library refuses everywhere else.
|
|
|
|
`_plan_covering` selects on content hash, so a mistyped `source_sha256`
|
|
matches nothing, every dropped file falls through to the one-concept rule,
|
|
and the run reports a perfectly ordinary success. The operator asked for
|
|
segmentation and got a flat bundle with no error to read. `vegnormal-okf`
|
|
is about to put an N500 corpus through this path; a silent zero there would
|
|
read as "the corpus has no concepts".
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT, source_sha256="0" * 64)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(tmp_path, plan=plan)
|
|
assert excinfo.value.code == "segmentation_plan_unmatched"
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_a_plan_matching_one_of_several_dropped_files_is_not_refused(tmp_path: Path) -> None:
|
|
"""The negative control for the check above.
|
|
|
|
A run may legitimately drop many documents while only one is plan-covered
|
|
-- that is the whole point of hash selection. A check that fired here would
|
|
have replaced a silent skip with a refusal of the normal case, so this test
|
|
is what keeps the new gate honest rather than merely loud.
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
drop(tmp_path / "round", "v720.md", "V720 Tunnel: egne krav.\n")
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT))
|
|
assert len(concepts(tmp_path / "bundle")) == 6
|
|
|
|
|
|
def test_an_unreadable_dropped_file_does_not_mask_an_unmatched_plan(tmp_path: Path) -> None:
|
|
"""A file the run could not read is not evidence that the plan matched.
|
|
|
|
Phase 1 skips a file it cannot read and records a `FailedFile`. If the
|
|
unmatched-plan check asked "was any file left unexamined?" instead of "was
|
|
a covering plan actually found?", an unreadable drop would suppress the
|
|
refusal and restore exactly the silence this closes. The plan here is
|
|
hashed over the unreadable file's OWN bytes, so it is the only drop that
|
|
could ever have matched.
|
|
"""
|
|
drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
other = "V720 Tunnel: egne krav.\n"
|
|
unreadable = drop(tmp_path / "round", "locked.md", other)
|
|
unreadable.chmod(0o000)
|
|
try:
|
|
# The premise, asserted rather than assumed: a test that ran as root
|
|
# would read the file fine and pass for the wrong reason.
|
|
with pytest.raises(OSError):
|
|
unreadable.read_bytes()
|
|
plan = build_plan(other.encode("utf-8"), other, paths=("krav/tunnel.md",))
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(tmp_path, plan=plan)
|
|
assert excinfo.value.code == "segmentation_plan_unmatched"
|
|
assert tree(tmp_path / "bundle") == {}
|
|
finally:
|
|
unreadable.chmod(0o644)
|
|
|
|
|
|
def test_a_matched_plan_with_a_refused_path_keeps_its_own_per_file_code(tmp_path: Path) -> None:
|
|
"""A matched plan with bad entries is not an unmatched plan.
|
|
|
|
Written because the first cut of the check got this backwards: it asked
|
|
whether any document reached the naming stage as covered, so a plan whose
|
|
hash matched but whose entry paths were then refused looked identical to a
|
|
plan that matched nothing. The operator would have been told to check a
|
|
hash that was already correct. Coverage is recorded at SELECTION.
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT, paths=("krav/" + "a" * 300 + ".md",))
|
|
result = run(tmp_path, plan=plan)
|
|
assert {entry.error.code for entry in result.failed} == {"inbox_slug_too_long"}
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_without_the_capability_an_unmatched_plan_is_still_the_earlier_refusal(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
"""Order matters: the profile check runs first and keeps its own code.
|
|
|
|
Both conditions hold in this call -- no capability AND no matching file --
|
|
and the operator's first problem is the profile, not the hash.
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT, source_sha256="0" * 64)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(tmp_path, plan=plan, profile=DEFAULT, values={})
|
|
assert excinfo.value.code == "segmentation_unsupported_profile"
|
|
|
|
|
|
# --- S5b: the cache key can actually fail ----------------------------------
|
|
#
|
|
# Both halves below were decorative before Step 11. The proposer hashed SOURCE
|
|
# BYTES only, so a converter that reshaped the extracted text left the hash
|
|
# identical and every offset moved under a key that still matched; and the run
|
|
# path passed `plan.extractor_version` straight back into `assert_plan_applies`,
|
|
# comparing the plan's value with itself. Two guards that could never fire, in
|
|
# the one place where a false pass produces a bundle nobody adjudicated and no
|
|
# downstream test can catch -- every span still lands on real text.
|
|
|
|
|
|
def test_a_plan_whose_extracted_text_hash_moved_is_refused(tmp_path: Path) -> None:
|
|
"""The signal a source-bytes hash cannot carry.
|
|
|
|
Same bytes on disk, same extractor id, same extractor version -- and a
|
|
different canonical text, which is what the offsets index. Simulated by
|
|
moving the hash rather than the converter, because the property under test
|
|
is that the component is COMPARED at all.
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT, text_sha256="0" * 64)
|
|
result = run(tmp_path, plan=plan)
|
|
assert {entry.error.code for entry in result.failed} == {"segmentation_extractor_mismatch"}
|
|
assert "text_sha256" in str(result.failed[0].error)
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_a_plan_whose_extractor_version_moved_is_refused(tmp_path: Path) -> None:
|
|
"""The half of S5b that compared a value with itself."""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT, extractor_version="not-the-one-that-ran")
|
|
result = run(tmp_path, plan=plan)
|
|
assert {entry.error.code for entry in result.failed} == {"segmentation_extractor_mismatch"}
|
|
assert "extractor_version" in str(result.failed[0].error)
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_the_observed_extractor_version_is_not_the_proposers_own(tmp_path: Path) -> None:
|
|
"""Defect (b): the proposer wrote ITS version into the extractor's field.
|
|
|
|
A stdlib row names this package's own literal because there is no third
|
|
party to name; a converter row names the pinned converter. What matters is
|
|
that the two are DIFFERENT values from different sources -- one tool
|
|
version standing in for both is exactly what made the field unable to move.
|
|
"""
|
|
assert observed_extractor_version("md") == STDLIB_EXTRACTOR_VERSION
|
|
assert observed_extractor_version("docx") != STDLIB_EXTRACTOR_VERSION
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
observed_extractor_version("nothing-registers-this")
|
|
assert excinfo.value.code == "segmentation_extractor_mismatch"
|
|
|
|
|
|
# --- one plan per document, so a corpus can be run at all ------------------
|
|
#
|
|
# `process_inbox` took ONE plan, so a round over a heterogeneous corpus with
|
|
# several segmented documents was not expressible: arm B cannot execute
|
|
# without this. The single-plan form stays exactly as it was -- the mapping
|
|
# arrives as a new keyword-only parameter with a default, so every existing
|
|
# call site is source-compatible and no golden moves.
|
|
#
|
|
# The mapping is keyed by SOURCE FILENAME because that is what an operator
|
|
# reads, but selection stays by CONTENT IDENTITY: `_plan_covering` matches on
|
|
# `source_sha256`, so a renamed file still finds its plan and a plan pointed at
|
|
# the wrong name still refuses rather than segmenting the wrong document.
|
|
|
|
OTHER = "A Innledning: hva dette dokumentet dekker.\nB Virkeomraade: hvilke anlegg det gjelder.\n"
|
|
|
|
OTHER_PATHS = ("annen/innledning.md", "annen/virkeomraade.md")
|
|
|
|
|
|
def test_two_documents_with_two_plans_both_segment_in_one_run(tmp_path: Path) -> None:
|
|
first = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
second = drop(tmp_path / "round", "n200.md", OTHER)
|
|
run(
|
|
tmp_path,
|
|
plan=None,
|
|
plans={
|
|
"n500.md": build_plan(first.read_bytes(), DOCUMENT),
|
|
"n200.md": build_plan(second.read_bytes(), OTHER, paths=OTHER_PATHS),
|
|
},
|
|
)
|
|
found = concepts(tmp_path / "bundle")
|
|
assert set(found) == set(PATHS) | set(OTHER_PATHS)
|
|
|
|
|
|
def test_a_mapping_selects_by_content_not_by_the_name_it_is_keyed_under(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
"""The key is for the operator; the hash is what decides.
|
|
|
|
Keyed under a name no dropped file carries, the plan still finds the bytes
|
|
it was adjudicated over. Anything else would make a rename silently produce
|
|
a flat bundle.
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
run(
|
|
tmp_path,
|
|
plan=None,
|
|
plans={"whatever-i-called-it.md": build_plan(source.read_bytes(), DOCUMENT)},
|
|
)
|
|
assert set(concepts(tmp_path / "bundle")) == set(PATHS)
|
|
|
|
|
|
def test_a_plan_in_the_mapping_that_matches_nothing_is_still_refused(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
"""The silent-skip refusal has to survive the wider input.
|
|
|
|
One plan matching is not enough: a corpus run where four of five plans
|
|
matched would otherwise report success over four segmented documents and
|
|
one flat one, which is exactly the silent zero this refusal exists for.
|
|
"""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(
|
|
tmp_path,
|
|
plan=None,
|
|
plans={
|
|
"n500.md": build_plan(source.read_bytes(), DOCUMENT),
|
|
"ghost.md": build_plan(source.read_bytes(), DOCUMENT, source_sha256="0" * 64),
|
|
},
|
|
)
|
|
assert excinfo.value.code == "segmentation_plan_unmatched"
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_passing_both_forms_at_once_is_refused(tmp_path: Path) -> None:
|
|
"""Two ways to say the same thing invite a silent disagreement."""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
plan = build_plan(source.read_bytes(), DOCUMENT)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(tmp_path, plan=plan, plans={"n500.md": plan})
|
|
assert excinfo.value.code == "segmentation_plan_invalid"
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_two_plans_claiming_the_same_bytes_are_refused(tmp_path: Path) -> None:
|
|
"""Which one would have segmented the document is not a coin toss."""
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
with pytest.raises(SegmentationError) as excinfo:
|
|
run(
|
|
tmp_path,
|
|
plan=None,
|
|
plans={
|
|
"a.md": build_plan(source.read_bytes(), DOCUMENT),
|
|
"b.md": build_plan(source.read_bytes(), DOCUMENT, paths=OTHER_PATHS + PATHS[2:]),
|
|
},
|
|
)
|
|
assert excinfo.value.code == "segmentation_plan_invalid"
|
|
assert tree(tmp_path / "bundle") == {}
|
|
|
|
|
|
def test_the_single_plan_form_is_unchanged(tmp_path: Path) -> None:
|
|
source = drop(tmp_path / "round", "n500.md", DOCUMENT)
|
|
run(tmp_path, plan=build_plan(source.read_bytes(), DOCUMENT))
|
|
assert set(concepts(tmp_path / "bundle")) == set(PATHS)
|