401 lines
17 KiB
Python
401 lines
17 KiB
Python
"""The segmentation plan: one document's split into many concepts, as data.
|
|
|
|
OKF v0.2 §2 defines a concept as "a single unit of knowledge within a bundle"
|
|
and a concept ID as the path of its file within the bundle. Neither ties a
|
|
concept to a source file, and Appendix A presents v0.1 -> v0.2 as a
|
|
de-monolithization. Door B nevertheless emitted exactly one flat concept per
|
|
dropped file, which is the form the SPEC names as the one being migrated away
|
|
from. No conformance test caught that and none could: §11 checks that every
|
|
non-reserved `.md` has parsable frontmatter with a non-empty `type`, so a
|
|
bundle of one giant concept is fully conformant. Conformance is the floor, not
|
|
the proof.
|
|
|
|
Splitting a document into units of knowledge is a JUDGEMENT, and this
|
|
library's run path promises zero model calls. The resolution is to make the
|
|
judgement once, write it down here as data, have a human adjudicate it, and
|
|
replay it deterministically forever after. A plan is therefore authored input,
|
|
never something this module infers: nothing below proposes a split, and the
|
|
proposer that does (`tools/okf_propose_segments.py`) lives outside the package
|
|
and marks every entry it emits as PROPOSED rather than adjudicated.
|
|
|
|
Two properties this module exists to protect:
|
|
|
|
1. **Offsets index the CANONICAL EXTRACTED TEXT, never the source bytes.** You
|
|
cannot slice a PDF's bytes and recover prose, and even a `.csv` is
|
|
re-rendered into a table before it becomes a concept body. A span is a
|
|
window on whatever `extract.extract_text` returned.
|
|
2. **Paths are normalised through the id grammar at entry.** macOS/APFS hands
|
|
filenames back DECOMPOSED, so the same visual path reduces two ways
|
|
depending on which normal form it arrived in. Normalising once, here,
|
|
is what keeps a concept ID from silently moving between rounds -- the one
|
|
failure that cannot be repaired after the fact, because consumers have
|
|
already linked to the old ID.
|
|
|
|
Pure: no filesystem, no bundle, no door, no model call, no network.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Mapping, Sequence
|
|
from dataclasses import dataclass, field
|
|
from pathlib import PurePosixPath
|
|
from typing import Any
|
|
|
|
from .errors import SegmentationError
|
|
from .materialize import reduce_to_id_grammar
|
|
|
|
#: The top-level keys a plan payload must carry. Every one is required: a plan
|
|
#: missing its extractor identity would still parse, and would then be replayed
|
|
#: against an extraction nobody checked it against.
|
|
PLAN_FIELDS = (
|
|
"version",
|
|
"source_sha256",
|
|
"extractor_id",
|
|
"extractor_version",
|
|
"adjudicated_at",
|
|
"entries",
|
|
)
|
|
|
|
#: The keys every entry must carry. `parent_id` and `derived` are optional --
|
|
#: a flat plan has no parents, and an entry adjudicated from scratch derived
|
|
#: nothing.
|
|
ENTRY_FIELDS = ("segment_id", "path", "title", "okf_type", "span", "ingested_at")
|
|
|
|
#: Path components refused outright, before the id grammar is consulted. An
|
|
#: empty component is a leading, trailing or doubled `/`; `.` and `..` are
|
|
#: traversal. Refused here rather than resolved, because a plan is authored and
|
|
#: an authored `..` is a mistake worth naming, not a path worth normalising.
|
|
FORBIDDEN_COMPONENTS = ("", ".", "..")
|
|
|
|
#: The components of the adjudication cache key, in the order
|
|
#: :func:`plan_cache_key` returns them. Named so a mismatch message can say
|
|
#: WHICH one moved -- that is what tells an operator whether to re-run the
|
|
#: proposer or re-adjudicate by hand.
|
|
CACHE_KEY_COMPONENTS = ("source_sha256", "extractor_id", "extractor_version")
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class SegmentEntry:
|
|
"""One concept a document expands into.
|
|
|
|
`span` is half-open over the canonical extracted text. `path` is
|
|
bundle-relative, `/`-separated and already normalised (see
|
|
:func:`normalize_segment_path`) -- the concept ID is this path minus the
|
|
suffix, so it is fixed the moment the plan is adjudicated.
|
|
"""
|
|
|
|
segment_id: str
|
|
path: str
|
|
title: str
|
|
okf_type: str
|
|
span: tuple[int, int]
|
|
ingested_at: str
|
|
parent_id: str | None = None
|
|
derived: frozenset[str] = field(default_factory=frozenset)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class SegmentationPlan:
|
|
"""An adjudicated split, keyed to the extraction it was adjudicated against.
|
|
|
|
The three extractor fields are not decoration. Source bytes cannot see an
|
|
extractor swap or a version bump, so `source_sha256` alone would still
|
|
match while every offset in `entries` had silently moved -- see
|
|
:func:`assert_plan_applies`.
|
|
"""
|
|
|
|
version: str
|
|
source_sha256: str
|
|
extractor_id: str
|
|
extractor_version: str
|
|
adjudicated_at: str
|
|
entries: tuple[SegmentEntry, ...]
|
|
|
|
|
|
def _require_str(payload: Mapping[str, Any], key: str, *, where: str) -> str:
|
|
if key not in payload:
|
|
raise SegmentationError(
|
|
f"{where} is missing the required field {key!r} — a plan is replayed "
|
|
"verbatim, so an absent field cannot be inferred",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
value = payload[key]
|
|
if not isinstance(value, str) or not value:
|
|
raise SegmentationError(
|
|
f"{where} field {key!r} must be a non-empty string, got {value!r}",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
return value
|
|
|
|
|
|
def normalize_segment_path(path: str, *, where: str) -> str:
|
|
"""The bundle-relative path an entry claims, reduced to the id grammar.
|
|
|
|
Every component is reduced separately, because reducing the joined string
|
|
would collapse the `/` separators into `-` and flatten the hierarchy the
|
|
plan exists to express. The last component's suffix is preserved rather
|
|
than reduced (`brannkonsept.md` must not become `brannkonsept-md`), which
|
|
is the same split Door B already makes on a dropped filename.
|
|
"""
|
|
if not isinstance(path, str) or not path:
|
|
raise SegmentationError(
|
|
f"{where} must carry a non-empty bundle-relative path, got {path!r}",
|
|
code="segmentation_path_invalid",
|
|
)
|
|
if "\\" in path:
|
|
raise SegmentationError(
|
|
f"{where} path {path!r} contains a backslash — paths are `/`-separated "
|
|
"and bundle-relative on every platform",
|
|
code="segmentation_path_invalid",
|
|
)
|
|
components = path.split("/")
|
|
forbidden = [item for item in components if item in FORBIDDEN_COMPONENTS]
|
|
if forbidden:
|
|
raise SegmentationError(
|
|
f"{where} path {path!r} is not bundle-relative — it is absolute, or it "
|
|
f"contains {', '.join(repr(item) for item in forbidden)}; refusing to "
|
|
"resolve traversal in authored data",
|
|
code="segmentation_path_invalid",
|
|
)
|
|
|
|
normalized: list[str] = []
|
|
last = len(components) - 1
|
|
for index, component in enumerate(components):
|
|
suffix = PurePosixPath(component).suffix if index == last else ""
|
|
stem = component[: len(component) - len(suffix)] if suffix else component
|
|
reduced = reduce_to_id_grammar(stem)
|
|
if not reduced:
|
|
raise SegmentationError(
|
|
f"{where} path {path!r} has a component {component!r} that reduces to "
|
|
"nothing under the id grammar ([a-z0-9][a-z0-9-]*) — refusing to "
|
|
"invent a directory name",
|
|
code="segmentation_path_invalid",
|
|
)
|
|
normalized.append(reduced + suffix.lower())
|
|
return "/".join(normalized)
|
|
|
|
|
|
def _parse_span(value: Any, *, where: str) -> tuple[int, int]:
|
|
if (
|
|
not isinstance(value, Sequence)
|
|
or isinstance(value, (str, bytes))
|
|
or len(value) != 2
|
|
or not all(isinstance(offset, int) for offset in value)
|
|
):
|
|
raise SegmentationError(
|
|
f"{where} span must be a two-item [start, end] of integer offsets into "
|
|
f"the canonical extracted text, got {value!r}",
|
|
code="segmentation_span_invalid",
|
|
)
|
|
start, end = int(value[0]), int(value[1])
|
|
if start < 0 or end <= start:
|
|
raise SegmentationError(
|
|
f"{where} span [{start}, {end}] is not a half-open range of non-negative "
|
|
"offsets with start < end — an empty or reversed span names no text",
|
|
code="segmentation_span_invalid",
|
|
)
|
|
return (start, end)
|
|
|
|
|
|
def _parse_derived(value: Any, *, where: str) -> frozenset[str]:
|
|
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
|
|
raise SegmentationError(
|
|
f"{where} field 'derived' must be a list of field names, got {value!r}",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
if not all(isinstance(name, str) and name for name in value):
|
|
raise SegmentationError(
|
|
f"{where} field 'derived' must hold non-empty field names, got {value!r}",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
return frozenset(value)
|
|
|
|
|
|
def _parse_entry(payload: Any, *, position: int) -> SegmentEntry:
|
|
where = f"segmentation entry {position}"
|
|
if not isinstance(payload, Mapping):
|
|
raise SegmentationError(
|
|
f"{where} must be a mapping, got {payload!r}",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
for key in ENTRY_FIELDS:
|
|
if key not in payload:
|
|
raise SegmentationError(
|
|
f"{where} is missing the required field {key!r} — a plan is replayed "
|
|
"verbatim, so an absent field cannot be inferred",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
|
|
segment_id = _require_str(payload, "segment_id", where=where)
|
|
where = f"segmentation entry {segment_id!r}"
|
|
parent_id = payload.get("parent_id")
|
|
if parent_id is not None and (not isinstance(parent_id, str) or not parent_id):
|
|
raise SegmentationError(
|
|
f"{where} field 'parent_id' must be a non-empty string or absent, got {parent_id!r}",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
return SegmentEntry(
|
|
segment_id=segment_id,
|
|
path=normalize_segment_path(payload["path"], where=where),
|
|
title=_require_str(payload, "title", where=where),
|
|
okf_type=_require_str(payload, "okf_type", where=where),
|
|
span=_parse_span(payload["span"], where=where),
|
|
ingested_at=_require_str(payload, "ingested_at", where=where),
|
|
parent_id=parent_id,
|
|
derived=_parse_derived(payload.get("derived", ()), where=where),
|
|
)
|
|
|
|
|
|
def parse_segmentation_plan(payload: Mapping[str, Any]) -> SegmentationPlan:
|
|
"""Validate an authored plan fail-fast, or refuse it with a typed code.
|
|
|
|
Fail-fast rather than best-effort: a plan is the record of a human
|
|
judgement, and a partially-honoured one would materialize a bundle nobody
|
|
adjudicated. Entries keep their authored order — the plan states the
|
|
document's own sequence, which no sort here could recover.
|
|
"""
|
|
if not isinstance(payload, Mapping):
|
|
raise SegmentationError(
|
|
f"a segmentation plan must be a mapping, got {payload!r}",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
for key in PLAN_FIELDS:
|
|
if key not in payload:
|
|
raise SegmentationError(
|
|
f"the segmentation plan is missing the required field {key!r} — a plan "
|
|
"is replayed verbatim, so an absent field cannot be inferred",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
|
|
raw_entries = payload["entries"]
|
|
if (
|
|
not isinstance(raw_entries, Sequence)
|
|
or isinstance(raw_entries, (str, bytes))
|
|
or not raw_entries
|
|
):
|
|
raise SegmentationError(
|
|
"a segmentation plan must name at least one entry — an empty plan would "
|
|
"silently persist nothing for a document that was dropped",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
|
|
entries = tuple(
|
|
_parse_entry(item, position=position) for position, item in enumerate(raw_entries)
|
|
)
|
|
|
|
seen_ids: set[str] = set()
|
|
for item in entries:
|
|
if item.segment_id in seen_ids:
|
|
raise SegmentationError(
|
|
f"two segmentation entries share the segment_id {item.segment_id!r} — "
|
|
"refusing to let plan order decide which one a parent points at",
|
|
code="segmentation_duplicate_id",
|
|
)
|
|
seen_ids.add(item.segment_id)
|
|
|
|
seen_paths: set[str] = set()
|
|
for item in entries:
|
|
if item.path in seen_paths:
|
|
raise SegmentationError(
|
|
f"two segmentation entries claim the path {item.path!r} after "
|
|
"normalisation — refusing to let one segment silently overwrite "
|
|
"the other",
|
|
code="segmentation_path_invalid",
|
|
)
|
|
seen_paths.add(item.path)
|
|
|
|
for item in entries:
|
|
if item.parent_id is not None and item.parent_id not in seen_ids:
|
|
raise SegmentationError(
|
|
f"segmentation entry {item.segment_id!r} names parent_id "
|
|
f"{item.parent_id!r}, which no entry in this plan carries — a "
|
|
"hierarchy is resolved inside one plan or not at all",
|
|
code="segmentation_plan_invalid",
|
|
)
|
|
|
|
return SegmentationPlan(
|
|
version=_require_str(payload, "version", where="the segmentation plan"),
|
|
source_sha256=_require_str(payload, "source_sha256", where="the segmentation plan"),
|
|
extractor_id=_require_str(payload, "extractor_id", where="the segmentation plan"),
|
|
extractor_version=_require_str(payload, "extractor_version", where="the segmentation plan"),
|
|
adjudicated_at=_require_str(payload, "adjudicated_at", where="the segmentation plan"),
|
|
entries=entries,
|
|
)
|
|
|
|
|
|
def plan_cache_key(plan: SegmentationPlan) -> tuple[str, str, str]:
|
|
"""The triple an adjudication is cached under: source AND extractor identity.
|
|
|
|
Not the hash alone. `source_sha256` answers "are these the same bytes?",
|
|
which is necessary and not sufficient: the offsets in a plan index the
|
|
canonical EXTRACTED text, and swapping the extractor or bumping its version
|
|
can re-shape that text while the source bytes are untouched. Keyed on the
|
|
hash alone, a stored adjudication would be replayed against text the
|
|
adjudicator never saw, and every span would land somewhere plausible and
|
|
wrong. This is design requirement S5b.
|
|
"""
|
|
return (plan.source_sha256, plan.extractor_id, plan.extractor_version)
|
|
|
|
|
|
def assert_plan_applies(
|
|
plan: SegmentationPlan,
|
|
*,
|
|
source_sha256: str,
|
|
extractor_id: str,
|
|
extractor_version: str,
|
|
) -> None:
|
|
"""Refuse loudly when a plan was adjudicated against a different extraction.
|
|
|
|
Loudly, and never by re-deriving: a silent fallback would turn "this plan
|
|
is stale" into "this bundle is subtly wrong", which no test downstream can
|
|
catch because every span still points at real text. The message names
|
|
which of the three components moved, because that is what tells the
|
|
operator whether to re-run the proposer or re-adjudicate by hand.
|
|
"""
|
|
observed = (source_sha256, extractor_id, extractor_version)
|
|
differing = [
|
|
f"{name}: plan {expected!r} != run {actual!r}"
|
|
for name, expected, actual in zip(CACHE_KEY_COMPONENTS, plan_cache_key(plan), observed)
|
|
if expected != actual
|
|
]
|
|
if differing:
|
|
raise SegmentationError(
|
|
"this segmentation plan was adjudicated against a different extraction "
|
|
f"({'; '.join(differing)}) — refusing to replay its offsets, which index "
|
|
"the canonical extracted text and would land on text no one adjudicated; "
|
|
"re-run the proposer and re-adjudicate",
|
|
code="segmentation_extractor_mismatch",
|
|
)
|
|
|
|
|
|
def slice_segments(text: str, plan: SegmentationPlan) -> tuple[tuple[SegmentEntry, str], ...]:
|
|
"""Pair every entry with the substring its declared span names, in plan order.
|
|
|
|
`text` is the CANONICAL EXTRACTED text -- whatever
|
|
:func:`llm_ingestion_okf.extract.extract_text` returned -- never the source
|
|
bytes. The distinction is not pedantry: a `.csv` is re-rendered as a table
|
|
and a `.pdf` has no sliceable prose at all, so an offset computed against
|
|
bytes would land on different characters and produce a concept body no one
|
|
adjudicated, with nothing failing.
|
|
|
|
Spans may OVERLAP and need not cover the whole text. Neither is asserted:
|
|
a preamble, a page header or a signature block is legitimately part of no
|
|
unit of knowledge, and forcing full coverage would make the adjudicator
|
|
invent a home for it. What is refused is a span reaching past the end --
|
|
that is not a judgement about the document but proof the plan was
|
|
adjudicated against a different extraction.
|
|
"""
|
|
limit = len(text)
|
|
sliced: list[tuple[SegmentEntry, str]] = []
|
|
for item in plan.entries:
|
|
start, end = item.span
|
|
if end > limit:
|
|
raise SegmentationError(
|
|
f"segmentation entry {item.segment_id!r} declares span [{start}, {end}] "
|
|
f"but the canonical extracted text is {limit} characters — the plan was "
|
|
"adjudicated against a different extraction; re-run the proposer and "
|
|
"re-adjudicate rather than truncating to fit",
|
|
code="segmentation_span_invalid",
|
|
)
|
|
sliced.append((item, text[start:end]))
|
|
return tuple(sliced)
|