"""The segmentation plan: one document's split into many concepts, as data. OKF v0.2 §2 defines a concept as "a single unit of knowledge within a bundle" and a concept ID as the path of its file within the bundle. Neither ties a concept to a source file, and Appendix A presents v0.1 -> v0.2 as a de-monolithization. Door B nevertheless emitted exactly one flat concept per dropped file, which is the form the SPEC names as the one being migrated away from. No conformance test caught that and none could: §11 checks that every non-reserved `.md` has parsable frontmatter with a non-empty `type`, so a bundle of one giant concept is fully conformant. Conformance is the floor, not the proof. Splitting a document into units of knowledge is a JUDGEMENT, and this library's run path promises zero model calls. The resolution is to make the judgement once, write it down here as data, have a human adjudicate it, and replay it deterministically forever after. A plan is therefore authored input, never something this module infers: nothing below proposes a split, and the proposer that does (`tools/okf_propose_segments.py`) lives outside the package and marks every entry it emits as PROPOSED rather than adjudicated. Two properties this module exists to protect: 1. **Offsets index the CANONICAL EXTRACTED TEXT, never the source bytes.** You cannot slice a PDF's bytes and recover prose, and even a `.csv` is re-rendered into a table before it becomes a concept body. A span is a window on whatever `extract.extract_text` returned. 2. **Paths are normalised through the id grammar at entry.** macOS/APFS hands filenames back DECOMPOSED, so the same visual path reduces two ways depending on which normal form it arrived in. Normalising once, here, is what keeps a concept ID from silently moving between rounds -- the one failure that cannot be repaired after the fact, because consumers have already linked to the old ID. Pure: no filesystem, no bundle, no door, no model call, no network. """ from __future__ import annotations from collections.abc import Mapping, Sequence from dataclasses import dataclass, field from pathlib import PurePosixPath from typing import Any from .errors import SegmentationError from .materialize import reduce_to_id_grammar #: The top-level keys a plan payload must carry. Every one is required: a plan #: missing its extractor identity would still parse, and would then be replayed #: against an extraction nobody checked it against. PLAN_FIELDS = ( "version", "source_sha256", "extractor_id", "extractor_version", "adjudicated_at", "entries", ) #: The keys every entry must carry. `parent_id` and `derived` are optional -- #: a flat plan has no parents, and an entry adjudicated from scratch derived #: nothing. ENTRY_FIELDS = ("segment_id", "path", "title", "okf_type", "span", "ingested_at") #: Path components refused outright, before the id grammar is consulted. An #: empty component is a leading, trailing or doubled `/`; `.` and `..` are #: traversal. Refused here rather than resolved, because a plan is authored and #: an authored `..` is a mistake worth naming, not a path worth normalising. FORBIDDEN_COMPONENTS = ("", ".", "..") #: The components of the adjudication cache key, in the order #: :func:`plan_cache_key` returns them. Named so a mismatch message can say #: WHICH one moved -- that is what tells an operator whether to re-run the #: proposer or re-adjudicate by hand. CACHE_KEY_COMPONENTS = ("source_sha256", "extractor_id", "extractor_version") @dataclass(frozen=True) class SegmentEntry: """One concept a document expands into. `span` is half-open over the canonical extracted text. `path` is bundle-relative, `/`-separated and already normalised (see :func:`normalize_segment_path`) -- the concept ID is this path minus the suffix, so it is fixed the moment the plan is adjudicated. """ segment_id: str path: str title: str okf_type: str span: tuple[int, int] ingested_at: str parent_id: str | None = None derived: frozenset[str] = field(default_factory=frozenset) @dataclass(frozen=True) class SegmentationPlan: """An adjudicated split, keyed to the extraction it was adjudicated against. The three extractor fields are not decoration. Source bytes cannot see an extractor swap or a version bump, so `source_sha256` alone would still match while every offset in `entries` had silently moved -- see :func:`assert_plan_applies`. """ version: str source_sha256: str extractor_id: str extractor_version: str adjudicated_at: str entries: tuple[SegmentEntry, ...] def _require_str(payload: Mapping[str, Any], key: str, *, where: str) -> str: if key not in payload: raise SegmentationError( f"{where} is missing the required field {key!r} — a plan is replayed " "verbatim, so an absent field cannot be inferred", code="segmentation_plan_invalid", ) value = payload[key] if not isinstance(value, str) or not value: raise SegmentationError( f"{where} field {key!r} must be a non-empty string, got {value!r}", code="segmentation_plan_invalid", ) return value def normalize_segment_path(path: str, *, where: str) -> str: """The bundle-relative path an entry claims, reduced to the id grammar. Every component is reduced separately, because reducing the joined string would collapse the `/` separators into `-` and flatten the hierarchy the plan exists to express. The last component's suffix is preserved rather than reduced (`brannkonsept.md` must not become `brannkonsept-md`), which is the same split Door B already makes on a dropped filename. """ if not isinstance(path, str) or not path: raise SegmentationError( f"{where} must carry a non-empty bundle-relative path, got {path!r}", code="segmentation_path_invalid", ) if "\\" in path: raise SegmentationError( f"{where} path {path!r} contains a backslash — paths are `/`-separated " "and bundle-relative on every platform", code="segmentation_path_invalid", ) components = path.split("/") forbidden = [item for item in components if item in FORBIDDEN_COMPONENTS] if forbidden: raise SegmentationError( f"{where} path {path!r} is not bundle-relative — it is absolute, or it " f"contains {', '.join(repr(item) for item in forbidden)}; refusing to " "resolve traversal in authored data", code="segmentation_path_invalid", ) normalized: list[str] = [] last = len(components) - 1 for index, component in enumerate(components): suffix = PurePosixPath(component).suffix if index == last else "" stem = component[: len(component) - len(suffix)] if suffix else component reduced = reduce_to_id_grammar(stem) if not reduced: raise SegmentationError( f"{where} path {path!r} has a component {component!r} that reduces to " "nothing under the id grammar ([a-z0-9][a-z0-9-]*) — refusing to " "invent a directory name", code="segmentation_path_invalid", ) normalized.append(reduced + suffix.lower()) return "/".join(normalized) def _parse_span(value: Any, *, where: str) -> tuple[int, int]: if ( not isinstance(value, Sequence) or isinstance(value, (str, bytes)) or len(value) != 2 or not all(isinstance(offset, int) for offset in value) ): raise SegmentationError( f"{where} span must be a two-item [start, end] of integer offsets into " f"the canonical extracted text, got {value!r}", code="segmentation_span_invalid", ) start, end = int(value[0]), int(value[1]) if start < 0 or end <= start: raise SegmentationError( f"{where} span [{start}, {end}] is not a half-open range of non-negative " "offsets with start < end — an empty or reversed span names no text", code="segmentation_span_invalid", ) return (start, end) def _parse_derived(value: Any, *, where: str) -> frozenset[str]: if not isinstance(value, Sequence) or isinstance(value, (str, bytes)): raise SegmentationError( f"{where} field 'derived' must be a list of field names, got {value!r}", code="segmentation_plan_invalid", ) if not all(isinstance(name, str) and name for name in value): raise SegmentationError( f"{where} field 'derived' must hold non-empty field names, got {value!r}", code="segmentation_plan_invalid", ) return frozenset(value) def _parse_entry(payload: Any, *, position: int) -> SegmentEntry: where = f"segmentation entry {position}" if not isinstance(payload, Mapping): raise SegmentationError( f"{where} must be a mapping, got {payload!r}", code="segmentation_plan_invalid", ) for key in ENTRY_FIELDS: if key not in payload: raise SegmentationError( f"{where} is missing the required field {key!r} — a plan is replayed " "verbatim, so an absent field cannot be inferred", code="segmentation_plan_invalid", ) segment_id = _require_str(payload, "segment_id", where=where) where = f"segmentation entry {segment_id!r}" parent_id = payload.get("parent_id") if parent_id is not None and (not isinstance(parent_id, str) or not parent_id): raise SegmentationError( f"{where} field 'parent_id' must be a non-empty string or absent, got {parent_id!r}", code="segmentation_plan_invalid", ) return SegmentEntry( segment_id=segment_id, path=normalize_segment_path(payload["path"], where=where), title=_require_str(payload, "title", where=where), okf_type=_require_str(payload, "okf_type", where=where), span=_parse_span(payload["span"], where=where), ingested_at=_require_str(payload, "ingested_at", where=where), parent_id=parent_id, derived=_parse_derived(payload.get("derived", ()), where=where), ) def parse_segmentation_plan(payload: Mapping[str, Any]) -> SegmentationPlan: """Validate an authored plan fail-fast, or refuse it with a typed code. Fail-fast rather than best-effort: a plan is the record of a human judgement, and a partially-honoured one would materialize a bundle nobody adjudicated. Entries keep their authored order — the plan states the document's own sequence, which no sort here could recover. """ if not isinstance(payload, Mapping): raise SegmentationError( f"a segmentation plan must be a mapping, got {payload!r}", code="segmentation_plan_invalid", ) for key in PLAN_FIELDS: if key not in payload: raise SegmentationError( f"the segmentation plan is missing the required field {key!r} — a plan " "is replayed verbatim, so an absent field cannot be inferred", code="segmentation_plan_invalid", ) raw_entries = payload["entries"] if ( not isinstance(raw_entries, Sequence) or isinstance(raw_entries, (str, bytes)) or not raw_entries ): raise SegmentationError( "a segmentation plan must name at least one entry — an empty plan would " "silently persist nothing for a document that was dropped", code="segmentation_plan_invalid", ) entries = tuple( _parse_entry(item, position=position) for position, item in enumerate(raw_entries) ) seen_ids: set[str] = set() for item in entries: if item.segment_id in seen_ids: raise SegmentationError( f"two segmentation entries share the segment_id {item.segment_id!r} — " "refusing to let plan order decide which one a parent points at", code="segmentation_duplicate_id", ) seen_ids.add(item.segment_id) seen_paths: set[str] = set() for item in entries: if item.path in seen_paths: raise SegmentationError( f"two segmentation entries claim the path {item.path!r} after " "normalisation — refusing to let one segment silently overwrite " "the other", code="segmentation_path_invalid", ) seen_paths.add(item.path) for item in entries: if item.parent_id is not None and item.parent_id not in seen_ids: raise SegmentationError( f"segmentation entry {item.segment_id!r} names parent_id " f"{item.parent_id!r}, which no entry in this plan carries — a " "hierarchy is resolved inside one plan or not at all", code="segmentation_plan_invalid", ) return SegmentationPlan( version=_require_str(payload, "version", where="the segmentation plan"), source_sha256=_require_str(payload, "source_sha256", where="the segmentation plan"), extractor_id=_require_str(payload, "extractor_id", where="the segmentation plan"), extractor_version=_require_str(payload, "extractor_version", where="the segmentation plan"), adjudicated_at=_require_str(payload, "adjudicated_at", where="the segmentation plan"), entries=entries, ) def plan_cache_key(plan: SegmentationPlan) -> tuple[str, str, str]: """The triple an adjudication is cached under: source AND extractor identity. Not the hash alone. `source_sha256` answers "are these the same bytes?", which is necessary and not sufficient: the offsets in a plan index the canonical EXTRACTED text, and swapping the extractor or bumping its version can re-shape that text while the source bytes are untouched. Keyed on the hash alone, a stored adjudication would be replayed against text the adjudicator never saw, and every span would land somewhere plausible and wrong. This is design requirement S5b. """ return (plan.source_sha256, plan.extractor_id, plan.extractor_version) def assert_plan_applies( plan: SegmentationPlan, *, source_sha256: str, extractor_id: str, extractor_version: str, ) -> None: """Refuse loudly when a plan was adjudicated against a different extraction. Loudly, and never by re-deriving: a silent fallback would turn "this plan is stale" into "this bundle is subtly wrong", which no test downstream can catch because every span still points at real text. The message names which of the three components moved, because that is what tells the operator whether to re-run the proposer or re-adjudicate by hand. """ observed = (source_sha256, extractor_id, extractor_version) differing = [ f"{name}: plan {expected!r} != run {actual!r}" for name, expected, actual in zip(CACHE_KEY_COMPONENTS, plan_cache_key(plan), observed) if expected != actual ] if differing: raise SegmentationError( "this segmentation plan was adjudicated against a different extraction " f"({'; '.join(differing)}) — refusing to replay its offsets, which index " "the canonical extracted text and would land on text no one adjudicated; " "re-run the proposer and re-adjudicate", code="segmentation_extractor_mismatch", ) def slice_segments(text: str, plan: SegmentationPlan) -> tuple[tuple[SegmentEntry, str], ...]: """Pair every entry with the substring its declared span names, in plan order. `text` is the CANONICAL EXTRACTED text -- whatever :func:`llm_ingestion_okf.extract.extract_text` returned -- never the source bytes. The distinction is not pedantry: a `.csv` is re-rendered as a table and a `.pdf` has no sliceable prose at all, so an offset computed against bytes would land on different characters and produce a concept body no one adjudicated, with nothing failing. Spans may OVERLAP and need not cover the whole text. Neither is asserted: a preamble, a page header or a signature block is legitimately part of no unit of knowledge, and forcing full coverage would make the adjudicator invent a home for it. What is refused is a span reaching past the end -- that is not a judgement about the document but proof the plan was adjudicated against a different extraction. """ limit = len(text) sliced: list[tuple[SegmentEntry, str]] = [] for item in plan.entries: start, end = item.span if end > limit: raise SegmentationError( f"segmentation entry {item.segment_id!r} declares span [{start}, {end}] " f"but the canonical extracted text is {limit} characters — the plan was " "adjudicated against a different extraction; re-run the proposer and " "re-adjudicate rather than truncating to fit", code="segmentation_span_invalid", ) sliced.append((item, text[start:end])) return tuple(sliced)