feat(segmentation): slice canonical extracted text by declared spans
This commit is contained in:
parent
832e541fb9
commit
499253e53b
2 changed files with 122 additions and 0 deletions
|
|
@ -27,6 +27,7 @@ from llm_ingestion_okf.segmentation import (
|
|||
assert_plan_applies,
|
||||
parse_segmentation_plan,
|
||||
plan_cache_key,
|
||||
slice_segments,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -343,3 +344,91 @@ def test_the_message_names_every_differing_component_not_just_the_first() -> Non
|
|||
assert "extractor_id" in str(error)
|
||||
assert "extractor_version" in str(error)
|
||||
assert "source_sha256" not in str(error)
|
||||
|
||||
|
||||
# --- slicing: spans are windows on the EXTRACTED text ----------------------
|
||||
#
|
||||
# Not on the source bytes. You cannot slice a PDF's bytes and recover prose,
|
||||
# and a `.csv` is re-rendered as a table before it becomes a concept body --
|
||||
# so an offset computed against bytes would land on different characters and
|
||||
# quietly produce a concept nobody adjudicated.
|
||||
|
||||
SECTIONS = (
|
||||
"0 Forord: bakgrunn for konseptet.\n" # 0..34
|
||||
"1 Brannkonsept: krav til seksjonering.\n" # 34..73
|
||||
"2 Roemning: to uavhengige veier.\n" # 73..106
|
||||
"3 Baereevne: R60 for hovedbaeresystem.\n" # 106..145
|
||||
"4 Slokkeanlegg: sprinkler i hele bygget.\n" # 145..186
|
||||
)
|
||||
|
||||
SPANS = ((0, 34), (34, 73), (73, 106), (106, 145), (145, 186))
|
||||
|
||||
|
||||
def five_section_plan(**overrides: Any) -> SegmentationPlan:
|
||||
spans = overrides.pop("spans", SPANS)
|
||||
return parse_segmentation_plan(
|
||||
plan(
|
||||
entries=[
|
||||
entry(segment_id=f"s{index}", path=f"krav/{index}-del.md", span=list(span))
|
||||
for index, span in enumerate(spans)
|
||||
],
|
||||
**overrides,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def test_five_declared_spans_yield_those_five_substrings_in_plan_order() -> None:
|
||||
sliced = slice_segments(SECTIONS, five_section_plan())
|
||||
assert len(sliced) == 5
|
||||
assert [item.segment_id for item, _ in sliced] == ["s0", "s1", "s2", "s3", "s4"]
|
||||
assert [body for _, body in sliced] == [SECTIONS[start:end] for start, end in SPANS]
|
||||
assert sliced[1][1].startswith("1 Brannkonsept")
|
||||
|
||||
|
||||
def test_moving_one_span_changes_exactly_that_one_body() -> None:
|
||||
# The S1 correspondence anchor: a concept's body is the span its OWN entry
|
||||
# declares. If a neighbour's body moved too, the split would be an artefact
|
||||
# of iteration rather than of the adjudication.
|
||||
def bodies_by_id(subject: SegmentationPlan) -> dict[str, str]:
|
||||
return {item.segment_id: body for item, body in slice_segments(SECTIONS, subject)}
|
||||
|
||||
baseline = bodies_by_id(five_section_plan())
|
||||
moved_spans = list(SPANS)
|
||||
moved_spans[2] = (73, 90)
|
||||
moved = bodies_by_id(five_section_plan(spans=tuple(moved_spans)))
|
||||
|
||||
assert set(moved) == set(baseline)
|
||||
assert {key for key in moved if moved[key] != baseline[key]} == {"s2"}
|
||||
|
||||
|
||||
def test_a_span_reaching_past_the_end_is_refused() -> None:
|
||||
subject = five_section_plan(spans=((0, 34), (34, len(SECTIONS) + 1)))
|
||||
with pytest.raises(SegmentationError) as excinfo:
|
||||
slice_segments(SECTIONS, subject)
|
||||
assert excinfo.value.code == "segmentation_span_invalid"
|
||||
assert str(len(SECTIONS)) in str(excinfo.value)
|
||||
|
||||
|
||||
def test_a_span_ending_exactly_at_the_end_is_permitted() -> None:
|
||||
sliced = slice_segments(SECTIONS, five_section_plan(spans=((145, len(SECTIONS)),)))
|
||||
assert sliced[0][1] == SECTIONS[145:]
|
||||
|
||||
|
||||
def test_an_uncovered_gap_in_the_middle_is_permitted() -> None:
|
||||
# A preamble, a page header or a signature block may belong to no concept.
|
||||
# Refusing a gap would force the adjudicator to invent a home for text that
|
||||
# is not a unit of knowledge.
|
||||
sliced = slice_segments(SECTIONS, five_section_plan(spans=((0, 34), (106, 145))))
|
||||
assert [body for _, body in sliced] == [SECTIONS[0:34], SECTIONS[106:145]]
|
||||
|
||||
|
||||
def test_overlapping_spans_are_permitted() -> None:
|
||||
sliced = slice_segments(SECTIONS, five_section_plan(spans=((0, 73), (34, 106))))
|
||||
assert sliced[0][1] == SECTIONS[0:73]
|
||||
assert sliced[1][1] == SECTIONS[34:106]
|
||||
|
||||
|
||||
def test_slicing_an_empty_text_with_any_span_is_refused() -> None:
|
||||
with pytest.raises(SegmentationError) as excinfo:
|
||||
slice_segments("", five_section_plan(spans=((0, 1),)))
|
||||
assert excinfo.value.code == "segmentation_span_invalid"
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue