feat(inbox): point every concept at the document it came from, with a locator per format
A concept named its source file by basename and, when segmented, carried a
`source_offset` into the text THIS LIBRARY extracted. Following that pointer
needed the corpus directory, the extractor and its exact transitive version --
none of which the bundle carries. Hand-walked on a real K2 concept: six steps,
four of them requiring knowledge from outside the bundle, to learn that a
requirement sits on pages 12-13 of a 20-page document.
The address is spec's: `sources: [{ resource, title }]`, where `resource` is
the dropped file's inbox-relative path (SPEC v0.2 5.1:303-306 -- "an absolute
URL, a bundle-relative path, or a path into a `references/` subdirectory").
The locator is ours, and it has to be: 5.1 has no field for a place within a
resource, and the pinned guard (1.3.0) rejects every route to putting one
inside a `sources` entry -- a non-allowlisted key by name, a nested flow list
as "scalar leaves only", and quoting as an unsupported form. So the locator is
top-level keys shaped like `source_offset`, and a path carrying a flow
terminator is refused fail-fast rather than mangled.
The unit table is built AT EXTRACTION, where the extracted text and the
original's structure are known to agree: pdf -> `source_pages` from
pdfplumber's own page numbers (a page that yielded no text does not renumber
the ones after it), xlsx -> `source_sheet` + `source_rows`, everything else ->
`source_lines`. `source_offset` stays.
Two measurements changed the design before it shipped. A `paragraphs` key for
docx would name a number the document does not have: `<w:p>` counts of
108/27/65/176/57 against converted-markdown lines of 75/33/67/144/63, not one
pair agreeing -- so the key is `source_lines` and says what it indexes. And an
empty spreadsheet row renders exactly like a table separator: the content-based
rule ate 8 empty rows on the K2 price sheet and reported its last row as 92
against a workbook that says 100. The separator is now found by position, and
`tomrad.xlsx` keeps that red.
One profile moves. `provenance` is a policy object, `None` everywhere but
`SEGMENTED_OKF_V0_2`; the other five shipped profiles are byte-identical.
K2 rebuilt from a frozen src copy: 629 concepts, 1108 files, name set identical,
0 ids moved, 479 files byte-identical, 629 changed and 0 lines removed anywhere.
629/629 now carry an address and a locator. New ref
`sha256-tree:665563a2f74423fcbcc8e4f0b0954ee73b73985ac0418de4f6987bd162a1f7c8`;
`2f82fcfe...` is stale. The pre-pass payload does not grow by one byte
(209 092 B before and after, 18 changed lines: the ref and eight per-concept
digests) -- because an excerpt carries the body, not the frontmatter, which is
also why the consumer still cannot cite "file X page 12" from a payload alone.
Report: docs/2026-09-08-proveniens-k2.md. 1339 tests, ruff and mypy clean.
Co-Authored-By: Claude <claude-opus-5>
This commit is contained in:
parent
d3bfe92acd
commit
b6a8c8bd89
16 changed files with 1301 additions and 26 deletions
|
|
@ -27,7 +27,7 @@ from dataclasses import dataclass, replace
|
|||
from pathlib import Path, PurePosixPath
|
||||
|
||||
from .errors import IngestError, MaterializationError, SegmentationError, SourceError
|
||||
from .extract import extract_text
|
||||
from .extract import SourceUnits, extract_text, source_units
|
||||
from .materialize import (
|
||||
_render_root_frontmatter,
|
||||
check_filename_length,
|
||||
|
|
@ -37,7 +37,7 @@ from .materialize import (
|
|||
validate_ingested_at,
|
||||
write_bytes,
|
||||
)
|
||||
from .profiles import DEFAULT, BundleProfile, IndexEntry
|
||||
from .profiles import DEFAULT, BundleProfile, IndexEntry, ProvenancePolicy
|
||||
from .segmentation import (
|
||||
SegmentationPlan,
|
||||
SegmentEntry,
|
||||
|
|
@ -111,6 +111,8 @@ def render_inbox_concept(
|
|||
structure: DocumentStructure | None = None,
|
||||
segment: SegmentEntry | None = None,
|
||||
bundle_id: str | None = None,
|
||||
units: SourceUnits | None = None,
|
||||
span: tuple[int, int] | None = None,
|
||||
) -> str:
|
||||
"""Frame extracted text as an inbox concept file with its provenance layer.
|
||||
|
||||
|
|
@ -119,6 +121,12 @@ def render_inbox_concept(
|
|||
`ingested_at`, on the reserved verdict layer, and on a title or
|
||||
`source_file` that would break an index link or inject frontmatter lines.
|
||||
|
||||
`units` and `span` carry the provenance locator and are read ONLY when the
|
||||
profile declares that capability. `span` defaults to the segment's own when
|
||||
the concept is segmented; a whole-document concept must supply it, because
|
||||
the text arriving here is the SANITIZED text and its length is not
|
||||
necessarily the extracted text's.
|
||||
|
||||
`segment` and `bundle_id` carry the 1-to-N identity layer and are read ONLY
|
||||
when the profile declares the segmentation capability. A concept the plan
|
||||
does not cover keeps today's rule verbatim, and the four shipped profiles
|
||||
|
|
@ -208,9 +216,78 @@ def render_inbox_concept(
|
|||
frontmatter["adjudicated_by"] = verdict.adjudicated_by
|
||||
frontmatter["adjudicated_at"] = verdict.adjudicated_at
|
||||
frontmatter["adjudication_dwell_s"] = str(verdict.adjudication_dwell_s)
|
||||
if profile.provenance is not None:
|
||||
# A segment's own span is the default, but only where the segment is
|
||||
# being READ -- `segmented` is the same discriminator the identity
|
||||
# layer above uses, so a profile with provenance and no segmentation
|
||||
# cannot silently locate by a span it is ignoring everywhere else.
|
||||
located = span
|
||||
if located is None and segmented:
|
||||
assert segment is not None
|
||||
located = segment.span
|
||||
frontmatter.update(
|
||||
_provenance_frontmatter(
|
||||
profile.provenance,
|
||||
source_file=source_file,
|
||||
units=units,
|
||||
span=located,
|
||||
)
|
||||
)
|
||||
return f"---\n{profile.frontmatter.emit(frontmatter)}\n---\n\n{_normalize_body(text)}"
|
||||
|
||||
|
||||
# The characters that would end a YAML flow mapping early, so a path carrying
|
||||
# one would produce a `sources` list that parses as something other than what
|
||||
# was written. The guard refuses a quoted scalar inside a flow mapping (1.3.0,
|
||||
# measured), so escaping is not on the table -- validation is.
|
||||
_FLOW_TERMINATORS = ",{}[]"
|
||||
|
||||
|
||||
def _provenance_frontmatter(
|
||||
policy: ProvenancePolicy,
|
||||
*,
|
||||
source_file: str,
|
||||
units: SourceUnits | None,
|
||||
span: tuple[int, int] | None,
|
||||
) -> dict[str, str]:
|
||||
"""The address, and the locator when one is available.
|
||||
|
||||
The address is written whether or not a locator is: `sources` answers
|
||||
"which document", the locator answers "where in it", and a consumer is owed
|
||||
the first even when the second cannot be computed.
|
||||
"""
|
||||
bad = [char for char in _FLOW_TERMINATORS if char in source_file]
|
||||
if bad:
|
||||
raise MaterializationError(
|
||||
f"source_file {source_file!r} contains {bad[0]!r}, which would end the "
|
||||
"`sources` flow mapping early; this profile writes an address a "
|
||||
"consumer can follow, and a path it cannot express is refused rather "
|
||||
"than mangled",
|
||||
code="inbox_source_file_unaddressable",
|
||||
)
|
||||
values = {
|
||||
policy.sources_key: (
|
||||
f"[{{ resource: {source_file}, title: {PurePosixPath(source_file).name} }}]"
|
||||
)
|
||||
}
|
||||
if units is None or span is None:
|
||||
return values
|
||||
first, last = units.covering(*span)
|
||||
if units.unit == "pages":
|
||||
values[policy.pages_key] = _render_flow_list([str(first), str(last)])
|
||||
elif units.unit == "rows":
|
||||
scopes = units.scopes_covering(*span)
|
||||
# A row number means nothing until a sheet is named, so a range that
|
||||
# crosses sheets gets neither key. An absence, never a first-sheet
|
||||
# guess: a guess here reads exactly like a fact.
|
||||
if len(scopes) == 1 and scopes[0] is not None:
|
||||
values[policy.sheet_key] = scopes[0]
|
||||
values[policy.rows_key] = _render_flow_list([str(first), str(last)])
|
||||
else:
|
||||
values[policy.lines_key] = _render_flow_list([str(first), str(last)])
|
||||
return values
|
||||
|
||||
|
||||
# --- the guard seam -------------------------------------------------------
|
||||
|
||||
# The guard's non-blocking floor. `Disposition` is a `str, Enum` in
|
||||
|
|
@ -542,6 +619,7 @@ def _render_segments(
|
|||
profile: BundleProfile,
|
||||
bundle_id: str,
|
||||
source_file: str,
|
||||
units: SourceUnits | None,
|
||||
) -> BlockedFile | None:
|
||||
"""Render every segment, or refuse the WHOLE document.
|
||||
|
||||
|
|
@ -607,6 +685,7 @@ def _render_segments(
|
|||
structure=structure,
|
||||
segment=entry,
|
||||
bundle_id=bundle_id,
|
||||
units=units,
|
||||
),
|
||||
decision.reasons,
|
||||
)
|
||||
|
|
@ -879,6 +958,16 @@ def process_inbox(
|
|||
source_bytes,
|
||||
renderer=_resolve_renderer(profile, path.name),
|
||||
)
|
||||
# Computed from the SAME text the plan's offsets index, so the
|
||||
# locator and the offset can never disagree about which rendering
|
||||
# they describe. `None` when the profile names no provenance:
|
||||
# building a unit table nobody writes would re-parse every PDF for
|
||||
# a key that is never emitted.
|
||||
units = (
|
||||
source_units(source_name(path), source_bytes, text)
|
||||
if profile.provenance is not None
|
||||
else None
|
||||
)
|
||||
covering = _plan_covering(plans, source_bytes)
|
||||
if covering is not None:
|
||||
blocked = _render_segments(
|
||||
|
|
@ -891,6 +980,7 @@ def process_inbox(
|
|||
profile=profile,
|
||||
bundle_id=(root_frontmatter_values or {})[_bundle_id_key(profile)],
|
||||
source_file=source_name(path),
|
||||
units=units,
|
||||
)
|
||||
if blocked is not None:
|
||||
if blocked.disposition == _DISPOSITION_QUARANTINE:
|
||||
|
|
@ -937,6 +1027,12 @@ def process_inbox(
|
|||
ingested_at=ingested_at,
|
||||
profile=profile,
|
||||
structure=structure,
|
||||
units=units,
|
||||
# The EXTRACTED text's span, never the sanitized
|
||||
# text's: the unit table indexes the former, and a
|
||||
# gate that removed a character would shift every
|
||||
# unit boundary after it.
|
||||
span=(0, len(text)),
|
||||
),
|
||||
decision.reasons,
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue