`_PANDOC_FORMATS` names the rows and no others: docx, xlsx, pptx, odt, rtf. `.html` stays on its stdlib extractor -- routing it through the converter would buy nothing and would add CVE-2025-51591 (SSRF via an iframe in HTML input), unpatched in every converter version. `.epub` is out on the "no gain" half of that. `_EVIDENCE` records what each row rests on, asserted in the suite rather than written in a comment: docx and xlsx are `measured`, and pptx, odt and rtf are `unmeasured` because the corpus contains ZERO files of those types. Three of five rows therefore leave this step working by construction and never checked against a document anyone wrote, and the assertion is what keeps that visible. Three converter arguments, all measured and none of them hygiene: `--eol=lf --wrap=none` because the defaults produce different bytes (max line length 75 against 447), and `-t markdown` never `-t plain` because plain destroys the headings the segment proposer reads -- 15 entries with two real headings become 13 with none. `_UNPARSED_OPTIONAL_EXTENSIONS` is now empty and kept rather than deleted: the branch still raises, and a future type arriving before its reader belongs there rather than in a new mechanism. This is what the first step was for -- both tests for `extractor_extra_missing` were repointed at the import probe before the set emptied under them. The converter call is isolated behind `_convert_bytes` so the seam's own logic is testable without the binary; the conversion itself is pinned by frozen-text fixtures in the next step. Checked live against a hand-laid docx through the real vendored binary: heading and body both survive. Suite 895 -> 908. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
378 lines
14 KiB
Python
378 lines
14 KiB
Python
"""Door B extraction registry: file bytes -> text, per file type (Phase 2 step 1).
|
|
|
|
Core stdlib extractors (md/txt/csv/json/html) plus the fail-fast gates: unknown
|
|
extension, the [extract]-gated binary types when the extra is absent, corrupt
|
|
(non-UTF-8) bytes, and an empty CSV. All file-type->text extraction lives HERE
|
|
(the guard is text-only); this step adds zero runtime dependency and no guard
|
|
call — it is pure, deterministic plumbing.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import importlib.util
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf import ExtractionError, ExtractionWarning, extract_text
|
|
|
|
FIXTURES = Path(__file__).parent / "fixtures"
|
|
|
|
# The parser-dependent tests below need the optional extra, so they are skipped
|
|
# without it — an optional extra that forced its parser on every test run would
|
|
# not be optional. The behaviour that must survive EITHER WAY is the rejection,
|
|
# and that one is asserted unconditionally above via the import probe.
|
|
requires_extract = pytest.mark.skipif(
|
|
importlib.util.find_spec("pdfplumber") is None,
|
|
reason="the optional [extract] extra is not installed",
|
|
)
|
|
|
|
# --- core extractors: happy path (a fixture per type) ---
|
|
|
|
|
|
def test_md_passthrough() -> None:
|
|
assert extract_text("note.md", b"# Title\n\nBody\n") == "# Title\n\nBody\n"
|
|
|
|
|
|
def test_txt_passthrough() -> None:
|
|
assert extract_text("note.txt", b"plain text") == "plain text"
|
|
|
|
|
|
def test_utf8_sig_bom_is_stripped() -> None:
|
|
# utf-8-sig so a BOM never leaks into the first character (baseline parity
|
|
# with Door A's read_csv).
|
|
assert extract_text("note.txt", b"\xef\xbb\xbfhello") == "hello"
|
|
|
|
|
|
def test_csv_renders_the_phase_1_markdown_table() -> None:
|
|
out = extract_text("data.csv", b"name,age\nAda,36\n")
|
|
assert out == "| name | age |\n| --- | --- |\n| Ada | 36 |\n"
|
|
|
|
|
|
def test_json_is_verbatim_inside_a_fenced_block() -> None:
|
|
out = extract_text("cfg.json", b'{"k": 1}\n')
|
|
assert out == '```\n{"k": 1}\n```\n'
|
|
|
|
|
|
def test_html_text_via_htmlparser() -> None:
|
|
# Block boundaries separate words; tags themselves contribute no text.
|
|
out = extract_text("page.html", b"<h1>Title</h1><p>Hello <b>world</b></p>")
|
|
assert out == "Title Hello world"
|
|
|
|
|
|
def test_html_skips_script_and_style() -> None:
|
|
html = b"<style>.x{color:red}</style><p>Keep</p><script>evil()</script>"
|
|
assert extract_text("page.html", html) == "Keep"
|
|
|
|
|
|
def test_htm_is_an_html_alias() -> None:
|
|
assert extract_text("page.htm", b"<p>hi</p>") == "hi"
|
|
|
|
|
|
def test_extension_dispatch_is_case_insensitive() -> None:
|
|
assert extract_text("NOTE.MD", b"x") == "x"
|
|
|
|
|
|
# --- fail-fast: unknown extension ---
|
|
|
|
|
|
def test_unknown_extension_fails_fast() -> None:
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("archive.zip", b"PK\x03\x04")
|
|
assert excinfo.value.code == "extractor_unknown"
|
|
|
|
|
|
def test_missing_extension_fails_fast() -> None:
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("README", b"x")
|
|
assert excinfo.value.code == "extractor_unknown"
|
|
|
|
|
|
# --- fail-fast: an [extract]-gated type without the extra installed ---
|
|
|
|
|
|
def test_optional_type_without_the_extra_fails_fast(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""`extractor_extra_missing` reached through the import probe, not a suffix set.
|
|
|
|
This test used to reach the code through a `.docx` or `.xlsx` filename,
|
|
which sat in `_UNPARSED_OPTIONAL_EXTENSIONS` because the extra shipped no
|
|
parser for those types. That set is on its way to empty: once those types gain a converter,
|
|
a membership test can no longer raise this code at all, and a test pinned
|
|
to it would go red for the right reason at the worst moment.
|
|
|
|
The import probe is the durable path — it is how the gate actually works
|
|
(`extract.py`'s `_extract_pdf`), and it stays reachable no matter how many
|
|
types gain parsers.
|
|
"""
|
|
monkeypatch.setitem(sys.modules, "pdfplumber", None)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("report.pdf", b"%PDF-1.4")
|
|
assert excinfo.value.code == "extractor_extra_missing"
|
|
# The error names the extra so the operator knows the remedy — never a
|
|
# silent skip, never a bundled parser in core.
|
|
assert "extract" in str(excinfo.value)
|
|
|
|
|
|
def test_pdf_without_the_extra_installed_keeps_the_same_rejection(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""The gate became an import probe; the rejection did not change.
|
|
|
|
Setting the module to None in sys.modules is what CPython treats as a
|
|
failed import, so this exercises the uninstalled path REGARDLESS of
|
|
whether the extra is installed in the running environment — a skip would
|
|
have preserved nothing on the machine where the parser is present.
|
|
"""
|
|
monkeypatch.setitem(sys.modules, "pdfplumber", None)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("doc.pdf", b"%PDF-1.4")
|
|
assert excinfo.value.code == "extractor_extra_missing"
|
|
assert "extract" in str(excinfo.value)
|
|
|
|
|
|
# --- office types: the table-driven converter seam ---
|
|
|
|
|
|
def test_every_office_row_names_its_reader() -> None:
|
|
"""The table is the contract; these are all the rows there are.
|
|
|
|
Written as an exact equality rather than a series of `in` checks: a row
|
|
added without a fixture and without an evidence class is the failure mode
|
|
this pins, and only equality catches an ADDITION.
|
|
"""
|
|
from llm_ingestion_okf.extract import _PANDOC_FORMATS
|
|
|
|
assert _PANDOC_FORMATS == {
|
|
".docx": "docx",
|
|
".xlsx": "xlsx",
|
|
".pptx": "pptx",
|
|
".odt": "odt",
|
|
".rtf": "rtf",
|
|
}
|
|
|
|
|
|
def test_html_and_epub_are_excluded_from_the_converter_on_purpose() -> None:
|
|
"""`.html` has a stdlib extractor; routing it through the converter would
|
|
buy nothing and would add CVE-2025-51591 (SSRF via an iframe in HTML
|
|
input), which is unpatched in every converter version. `.epub` is out for
|
|
the same "no gain" half of that reason.
|
|
"""
|
|
from llm_ingestion_okf.extract import _CORE_EXTRACTORS, _PANDOC_FORMATS
|
|
|
|
assert ".html" not in _PANDOC_FORMATS
|
|
assert ".epub" not in _PANDOC_FORMATS
|
|
assert ".html" in _CORE_EXTRACTORS
|
|
|
|
|
|
def test_evidence_class_is_asserted_not_commented() -> None:
|
|
"""Which rows were measured is a fact about this work, not a footnote.
|
|
|
|
`.pptx`, `.odt` and `.rtf` have denominator ZERO in the corpus this arm was
|
|
measured on. A comment saying so rots; an assertion that names them keeps
|
|
an unmeasured row from quietly presenting as a supported one.
|
|
"""
|
|
from llm_ingestion_okf.extract import _EVIDENCE, _PANDOC_FORMATS
|
|
|
|
assert set(_EVIDENCE) == set(_PANDOC_FORMATS), "every row needs an evidence class"
|
|
assert {s for s, e in _EVIDENCE.items() if e == "measured"} == {".docx", ".xlsx"}
|
|
assert {s for s, e in _EVIDENCE.items() if e == "unmeasured"} == {
|
|
".pptx",
|
|
".odt",
|
|
".rtf",
|
|
}
|
|
|
|
|
|
def test_the_unparsed_set_is_empty_now_that_every_row_has_a_reader() -> None:
|
|
"""The membership branch that used to raise `extractor_extra_missing` has
|
|
no members left. This is why both tests for that code were repointed at the
|
|
import probe before this step landed.
|
|
"""
|
|
from llm_ingestion_okf.extract import _UNPARSED_OPTIONAL_EXTENSIONS
|
|
|
|
assert _UNPARSED_OPTIONAL_EXTENSIONS == frozenset()
|
|
|
|
|
|
def test_the_converter_is_called_with_the_load_bearing_arguments(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""`--eol=lf --wrap=none` and the markdown writer, all three measured.
|
|
|
|
Not hygiene, and not style. The defaults produce DIFFERENT BYTES (maximum
|
|
line length 75 against 447), which a byte-pinned golden would register as a
|
|
change nobody made. And `-t plain` destroys the headings the segment
|
|
proposer reads: measured, a document yielding 15 entries including two real
|
|
headings yields 13 with none once the writer is `plain`.
|
|
"""
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
seen: dict[str, object] = {}
|
|
|
|
def fake_convert(source: object, to: str, format: str, extra_args: object) -> str:
|
|
seen.update(to=to, format=format, extra_args=list(extra_args)) # type: ignore[call-overload]
|
|
return "# Heading\n\nBody."
|
|
|
|
monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert)
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text("note.docx", b"PK\x03\x04")
|
|
|
|
assert text == "# Heading\n\nBody."
|
|
assert seen["to"] == "markdown"
|
|
assert seen["format"] == "docx"
|
|
assert "--eol=lf" in seen["extra_args"] # type: ignore[operator]
|
|
assert "--wrap=none" in seen["extra_args"] # type: ignore[operator]
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("name", "reader"),
|
|
[
|
|
("a.docx", "docx"),
|
|
("b.xlsx", "xlsx"),
|
|
("c.pptx", "pptx"),
|
|
("d.odt", "odt"),
|
|
("e.rtf", "rtf"),
|
|
],
|
|
)
|
|
def test_each_row_dispatches_to_its_reader(
|
|
name: str, reader: str, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
seen: dict[str, object] = {}
|
|
|
|
def fake_convert(source: object, to: str, format: str, extra_args: object) -> str:
|
|
seen["format"] = format
|
|
return "text"
|
|
|
|
monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert)
|
|
with pytest.warns(ExtractionWarning):
|
|
extract_text(name, b"bytes")
|
|
assert seen["format"] == reader
|
|
|
|
|
|
def test_an_empty_conversion_is_refused_not_persisted(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""Same reason as `extractor_empty_pdf`: an empty concept is the silent
|
|
skip this registry exists to prevent."""
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
monkeypatch.setattr(
|
|
extract_module,
|
|
"_convert_bytes",
|
|
lambda source, to, format, extra_args: " \n\t ",
|
|
)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("empty.docx", b"bytes")
|
|
assert excinfo.value.code == "extractor_empty_conversion"
|
|
|
|
|
|
def test_a_converter_failure_is_wrapped_never_leaked(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
def boom(source: object, to: str, format: str, extra_args: object) -> str:
|
|
raise RuntimeError("pandoc said no")
|
|
|
|
monkeypatch.setattr(extract_module, "_convert_bytes", boom)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("bad.docx", b"bytes")
|
|
assert excinfo.value.code == "extractor_convert_error"
|
|
assert "pandoc said no" in str(excinfo.value)
|
|
|
|
|
|
def test_office_conversion_warns_that_it_is_lossy(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""Drawn content does not survive extraction here either, and the warning
|
|
is emitted after the parse -- a run that produced no text has nothing to be
|
|
lossy about."""
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
monkeypatch.setattr(
|
|
extract_module,
|
|
"_convert_bytes",
|
|
lambda source, to, format, extra_args: "body",
|
|
)
|
|
with pytest.warns(ExtractionWarning):
|
|
extract_text("note.docx", b"bytes")
|
|
|
|
|
|
# --- pdf: the [extract] parser (pdfplumber) ---
|
|
|
|
# The expected text is frozen against a committed fixture ON PURPOSE. Extraction
|
|
# is deterministic within a parser version and NOT guaranteed across one
|
|
# (pdfplumber pins pdfminer.six==20260107 exactly; pdfminer.six ships
|
|
# date-stamped releases with no stability contract). This literal is what makes
|
|
# a parser upgrade break something visible instead of drifting silently.
|
|
KRAV_TEXT = "Krav til helning på utkilingen\n60 og 70 1:15"
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_extracts_text_with_label_and_value_on_one_line() -> None:
|
|
data = (FIXTURES / "two-line-krav.pdf").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
assert extract_text("krav.pdf", data) == KRAV_TEXT
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_extraction_is_byte_stable_across_calls() -> None:
|
|
"""Determinism is a promise this library already makes; hold it here too."""
|
|
data = (FIXTURES / "two-line-krav.pdf").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
first = extract_text("krav.pdf", data)
|
|
second = extract_text("krav.pdf", data)
|
|
assert first == second
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_extraction_warns_that_drawn_content_is_not_recovered() -> None:
|
|
"""Figures are vector drawings: only the caption survives leg 2.
|
|
|
|
Categorically true of text extraction, so it is stated as a warning on
|
|
every PDF rather than guessed at per document — detecting "is there a
|
|
figure here" would be exactly the layout heuristic this order declined.
|
|
"""
|
|
data = (FIXTURES / "two-line-krav.pdf").read_bytes()
|
|
with pytest.warns(ExtractionWarning, match="figures"):
|
|
extract_text("krav.pdf", data)
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_with_no_text_layer_fails_fast() -> None:
|
|
"""A scanned PDF yields nothing; an empty concept would be a silent skip."""
|
|
data = (FIXTURES / "no-text-layer.pdf").read_bytes()
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("scan.pdf", data)
|
|
assert excinfo.value.code == "extractor_empty_pdf"
|
|
|
|
|
|
@requires_extract
|
|
def test_corrupt_pdf_fails_fast_typed() -> None:
|
|
"""Never a leaked pdfminer exception — the "always typed" doctrine holds."""
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("broken.pdf", b"not a pdf at all")
|
|
assert excinfo.value.code == "extractor_pdf_error"
|
|
|
|
|
|
# --- fail-fast: corrupt (non-UTF-8) bytes on a text type ---
|
|
|
|
|
|
def test_invalid_utf8_fails_fast_typed() -> None:
|
|
# Never a leaked UnicodeDecodeError — the "always typed" doctrine holds for
|
|
# Door B too (cf. Door A wrapping path ValueError as SourceError).
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("note.txt", b"\xffbad")
|
|
assert excinfo.value.code == "extractor_decode_error"
|
|
|
|
|
|
# --- fail-fast: empty CSV (no header row) ---
|
|
|
|
|
|
def test_empty_csv_fails_fast() -> None:
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("empty.csv", b"")
|
|
assert excinfo.value.code == "extractor_empty_csv"
|