"""Door B extraction registry: file bytes -> text, per file type (Phase 2 step 1). Core stdlib extractors (md/txt/csv/json/html) plus the fail-fast gates: unknown extension, the [extract]-gated binary types when the extra is absent, corrupt (non-UTF-8) bytes, and an empty CSV. All file-type->text extraction lives HERE (the guard is text-only); this step adds zero runtime dependency and no guard call — it is pure, deterministic plumbing. """ from __future__ import annotations import importlib.util import sys from pathlib import Path import pytest from llm_ingestion_okf import ExtractionError, ExtractionWarning, extract_text FIXTURES = Path(__file__).parent / "fixtures" # The parser-dependent tests below need the optional extra, so they are skipped # without it — an optional extra that forced its parser on every test run would # not be optional. The behaviour that must survive EITHER WAY is the rejection, # and that one is asserted unconditionally above via the import probe. requires_extract = pytest.mark.skipif( importlib.util.find_spec("pdfplumber") is None, reason="the optional [extract] extra is not installed", ) # --- core extractors: happy path (a fixture per type) --- def test_md_passthrough() -> None: assert extract_text("note.md", b"# Title\n\nBody\n") == "# Title\n\nBody\n" def test_txt_passthrough() -> None: assert extract_text("note.txt", b"plain text") == "plain text" def test_utf8_sig_bom_is_stripped() -> None: # utf-8-sig so a BOM never leaks into the first character (baseline parity # with Door A's read_csv). assert extract_text("note.txt", b"\xef\xbb\xbfhello") == "hello" def test_csv_renders_the_phase_1_markdown_table() -> None: out = extract_text("data.csv", b"name,age\nAda,36\n") assert out == "| name | age |\n| --- | --- |\n| Ada | 36 |\n" def test_json_is_verbatim_inside_a_fenced_block() -> None: out = extract_text("cfg.json", b'{"k": 1}\n') assert out == '```\n{"k": 1}\n```\n' def test_html_text_via_htmlparser() -> None: # Block boundaries separate words; tags themselves contribute no text. out = extract_text("page.html", b"
Hello world
") assert out == "Title Hello world" def test_html_skips_script_and_style() -> None: html = b"Keep
" assert extract_text("page.html", html) == "Keep" def test_htm_is_an_html_alias() -> None: assert extract_text("page.htm", b"hi
") == "hi" def test_extension_dispatch_is_case_insensitive() -> None: assert extract_text("NOTE.MD", b"x") == "x" # --- fail-fast: unknown extension --- def test_unknown_extension_fails_fast() -> None: with pytest.raises(ExtractionError) as excinfo: extract_text("archive.zip", b"PK\x03\x04") assert excinfo.value.code == "extractor_unknown" def test_missing_extension_fails_fast() -> None: with pytest.raises(ExtractionError) as excinfo: extract_text("README", b"x") assert excinfo.value.code == "extractor_unknown" # --- fail-fast: an [extract]-gated type without the extra installed --- def test_optional_type_without_the_extra_fails_fast( monkeypatch: pytest.MonkeyPatch, ) -> None: """`extractor_extra_missing` reached through the import probe, not a suffix set. This test used to reach the code through a `.docx` or `.xlsx` filename, which sat in `_UNPARSED_OPTIONAL_EXTENSIONS` because the extra shipped no parser for those types. That set is on its way to empty: once those types gain a converter, a membership test can no longer raise this code at all, and a test pinned to it would go red for the right reason at the worst moment. The import probe is the durable path — it is how the gate actually works (`extract.py`'s `_extract_pdf`), and it stays reachable no matter how many types gain parsers. """ monkeypatch.setitem(sys.modules, "pdfplumber", None) with pytest.raises(ExtractionError) as excinfo: extract_text("report.pdf", b"%PDF-1.4") assert excinfo.value.code == "extractor_extra_missing" # The error names the extra so the operator knows the remedy — never a # silent skip, never a bundled parser in core. assert "extract" in str(excinfo.value) def test_pdf_without_the_extra_installed_keeps_the_same_rejection( monkeypatch: pytest.MonkeyPatch, ) -> None: """The gate became an import probe; the rejection did not change. Setting the module to None in sys.modules is what CPython treats as a failed import, so this exercises the uninstalled path REGARDLESS of whether the extra is installed in the running environment — a skip would have preserved nothing on the machine where the parser is present. """ monkeypatch.setitem(sys.modules, "pdfplumber", None) with pytest.raises(ExtractionError) as excinfo: extract_text("doc.pdf", b"%PDF-1.4") assert excinfo.value.code == "extractor_extra_missing" assert "extract" in str(excinfo.value) # --- office types: the table-driven converter seam --- def test_every_office_row_names_its_reader() -> None: """The table is the contract; these are all the rows there are. Written as an exact equality rather than a series of `in` checks: a row added without a fixture and without an evidence class is the failure mode this pins, and only equality catches an ADDITION. """ from llm_ingestion_okf.extract import _PANDOC_FORMATS assert _PANDOC_FORMATS == { ".docx": "docx", ".xlsx": "xlsx", ".pptx": "pptx", ".odt": "odt", ".rtf": "rtf", } def test_html_and_epub_are_excluded_from_the_converter_on_purpose() -> None: """`.html` has a stdlib extractor; routing it through the converter would buy nothing and would add CVE-2025-51591 (SSRF via an iframe in HTML input), which is unpatched in every converter version. `.epub` is out for the same "no gain" half of that reason. """ from llm_ingestion_okf.extract import _CORE_EXTRACTORS, _PANDOC_FORMATS assert ".html" not in _PANDOC_FORMATS assert ".epub" not in _PANDOC_FORMATS assert ".html" in _CORE_EXTRACTORS def test_evidence_class_is_asserted_not_commented() -> None: """Which rows were measured is a fact about this work, not a footnote. `.pptx`, `.odt` and `.rtf` have denominator ZERO in the corpus this arm was measured on. A comment saying so rots; an assertion that names them keeps an unmeasured row from quietly presenting as a supported one. """ from llm_ingestion_okf.extract import _EVIDENCE, _PANDOC_FORMATS assert set(_EVIDENCE) == set(_PANDOC_FORMATS), "every row needs an evidence class" assert {s for s, e in _EVIDENCE.items() if e == "measured"} == {".docx", ".xlsx"} assert {s for s, e in _EVIDENCE.items() if e == "unmeasured"} == { ".pptx", ".odt", ".rtf", } def test_the_unparsed_set_is_empty_now_that_every_row_has_a_reader() -> None: """The membership branch that used to raise `extractor_extra_missing` has no members left. This is why both tests for that code were repointed at the import probe before this step landed. """ from llm_ingestion_okf.extract import _UNPARSED_OPTIONAL_EXTENSIONS assert _UNPARSED_OPTIONAL_EXTENSIONS == frozenset() def test_the_converter_is_called_with_the_load_bearing_arguments( monkeypatch: pytest.MonkeyPatch, ) -> None: """`--eol=lf --wrap=none` and the markdown writer, all three measured. Not hygiene, and not style. The defaults produce DIFFERENT BYTES (maximum line length 75 against 447), which a byte-pinned golden would register as a change nobody made. And `-t plain` destroys the headings the segment proposer reads: measured, a document yielding 15 entries including two real headings yields 13 with none once the writer is `plain`. """ import llm_ingestion_okf.extract as extract_module seen: dict[str, object] = {} def fake_convert(source: object, to: str, format: str, extra_args: object) -> str: seen.update(to=to, format=format, extra_args=list(extra_args)) # type: ignore[call-overload] return "# Heading\n\nBody." monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert) with pytest.warns(ExtractionWarning): text = extract_text("note.docx", b"PK\x03\x04") assert text == "# Heading\n\nBody." assert seen["to"] == "markdown" assert seen["format"] == "docx" assert "--eol=lf" in seen["extra_args"] # type: ignore[operator] assert "--wrap=none" in seen["extra_args"] # type: ignore[operator] @pytest.mark.parametrize( ("name", "reader"), [ ("a.docx", "docx"), ("b.xlsx", "xlsx"), ("c.pptx", "pptx"), ("d.odt", "odt"), ("e.rtf", "rtf"), ], ) def test_each_row_dispatches_to_its_reader( name: str, reader: str, monkeypatch: pytest.MonkeyPatch ) -> None: import llm_ingestion_okf.extract as extract_module seen: dict[str, object] = {} def fake_convert(source: object, to: str, format: str, extra_args: object) -> str: seen["format"] = format return "text" monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert) with pytest.warns(ExtractionWarning): extract_text(name, b"bytes") assert seen["format"] == reader def test_an_empty_conversion_is_refused_not_persisted( monkeypatch: pytest.MonkeyPatch, ) -> None: """Same reason as `extractor_empty_pdf`: an empty concept is the silent skip this registry exists to prevent.""" import llm_ingestion_okf.extract as extract_module monkeypatch.setattr( extract_module, "_convert_bytes", lambda source, to, format, extra_args: " \n\t ", ) with pytest.raises(ExtractionError) as excinfo: extract_text("empty.docx", b"bytes") assert excinfo.value.code == "extractor_empty_conversion" def test_a_converter_failure_is_wrapped_never_leaked( monkeypatch: pytest.MonkeyPatch, ) -> None: import llm_ingestion_okf.extract as extract_module def boom(source: object, to: str, format: str, extra_args: object) -> str: raise RuntimeError("pandoc said no") monkeypatch.setattr(extract_module, "_convert_bytes", boom) with pytest.raises(ExtractionError) as excinfo: extract_text("bad.docx", b"bytes") assert excinfo.value.code == "extractor_convert_error" assert "pandoc said no" in str(excinfo.value) def test_office_conversion_warns_that_it_is_lossy( monkeypatch: pytest.MonkeyPatch, ) -> None: """Drawn content does not survive extraction here either, and the warning is emitted after the parse -- a run that produced no text has nothing to be lossy about.""" import llm_ingestion_okf.extract as extract_module monkeypatch.setattr( extract_module, "_convert_bytes", lambda source, to, format, extra_args: "body", ) with pytest.warns(ExtractionWarning): extract_text("note.docx", b"bytes") # --- pdf: the [extract] parser (pdfplumber) --- # The expected text is frozen against a committed fixture ON PURPOSE. Extraction # is deterministic within a parser version and NOT guaranteed across one # (pdfplumber pins pdfminer.six==20260107 exactly; pdfminer.six ships # date-stamped releases with no stability contract). This literal is what makes # a parser upgrade break something visible instead of drifting silently. KRAV_TEXT = "Krav til helning på utkilingen\n60 og 70 1:15" # The office fixtures are frozen the same way, and against a NAMED converter # version -- a frozen literal means nothing without one, because the thing it # pins is "this converter, on this input, produces these bytes". `_pandoc.py` # refuses any other version, so the two pins hold each other up. requires_pandoc = pytest.mark.skipif( importlib.util.find_spec("pypandoc") is None, reason="the optional [extract] extra is not installed", ) # docx: a heading and one requirement row with label and value on the SAME # line, mirroring the property the PDF fixture pins. DOCX_TEXT = "# Krav til helning\n\n60 og 70 1:15" # xlsx: the sheet name becomes a heading and the rows become a table. The # label/value pairing survives on one row, which is the property that matters. XLSX_TEXT = ( "## Krav {#sheet-1}\n\n Krav til helning \n" " ------------------ ------\n 60 og 70 1:15" ) # The negative control, committed rather than described: the SAME document # without `word/styles.xml`. The body survives and the heading marker does not. DOCX_NO_STYLES_TEXT = "Krav til helning\n\n60 og 70 1:15" @requires_pandoc def test_docx_extracts_to_its_frozen_text() -> None: data = (FIXTURES / "two-line-krav.docx").read_bytes() with pytest.warns(ExtractionWarning): assert extract_text("krav.docx", data) == DOCX_TEXT @requires_pandoc def test_xlsx_extracts_to_its_frozen_text() -> None: data = (FIXTURES / "two-line-krav.xlsx").read_bytes() with pytest.warns(ExtractionWarning): assert extract_text("krav.xlsx", data) == XLSX_TEXT @requires_pandoc def test_a_docx_without_a_styles_part_loses_its_heading() -> None: """The negative control for the fixture policy, run rather than asserted. `word/styles.xml` is what makes the converter see a heading. Without it the same document extracts as flat prose -- so a fixture built WITHOUT that part would pin the body and pin nothing at all about structure, while looking exactly as convincing. Structure is the half the segment proposer reads, which is why this is a committed fixture and not a sentence in a README. """ data = (FIXTURES / "no-styles-krav.docx").read_bytes() with pytest.warns(ExtractionWarning): text = extract_text("krav.docx", data) assert text == DOCX_NO_STYLES_TEXT assert not text.startswith("#"), "the heading marker must be absent" assert DOCX_TEXT.startswith("#"), "and present in the fixture that has styles" @requires_pandoc def test_the_frozen_office_text_is_pinned_to_a_named_converter_version() -> None: """A frozen literal without a named version pins nothing. If the converter version ever moves, these literals must be re-measured rather than trusted -- so the version is asserted right where they live. """ from llm_ingestion_okf._pandoc import PANDOC_VERSION, resolve_pandoc assert PANDOC_VERSION == "3.9" assert resolve_pandoc().is_file() @requires_pandoc def test_office_extraction_is_byte_stable_across_calls() -> None: data = (FIXTURES / "two-line-krav.docx").read_bytes() with pytest.warns(ExtractionWarning): first = extract_text("krav.docx", data) with pytest.warns(ExtractionWarning): second = extract_text("krav.docx", data) assert first == second @requires_extract def test_pdf_extracts_text_with_label_and_value_on_one_line() -> None: data = (FIXTURES / "two-line-krav.pdf").read_bytes() with pytest.warns(ExtractionWarning): assert extract_text("krav.pdf", data) == KRAV_TEXT @requires_extract def test_pdf_extraction_is_byte_stable_across_calls() -> None: """Determinism is a promise this library already makes; hold it here too.""" data = (FIXTURES / "two-line-krav.pdf").read_bytes() with pytest.warns(ExtractionWarning): first = extract_text("krav.pdf", data) second = extract_text("krav.pdf", data) assert first == second @requires_extract def test_pdf_extraction_warns_that_drawn_content_is_not_recovered() -> None: """Figures are vector drawings: only the caption survives leg 2. Categorically true of text extraction, so it is stated as a warning on every PDF rather than guessed at per document — detecting "is there a figure here" would be exactly the layout heuristic this order declined. """ data = (FIXTURES / "two-line-krav.pdf").read_bytes() with pytest.warns(ExtractionWarning, match="figures"): extract_text("krav.pdf", data) @requires_extract def test_pdf_with_no_text_layer_fails_fast() -> None: """A scanned PDF yields nothing; an empty concept would be a silent skip.""" data = (FIXTURES / "no-text-layer.pdf").read_bytes() with pytest.raises(ExtractionError) as excinfo: extract_text("scan.pdf", data) assert excinfo.value.code == "extractor_empty_pdf" @requires_extract def test_corrupt_pdf_fails_fast_typed() -> None: """Never a leaked pdfminer exception — the "always typed" doctrine holds.""" with pytest.raises(ExtractionError) as excinfo: extract_text("broken.pdf", b"not a pdf at all") assert excinfo.value.code == "extractor_pdf_error" # --- fail-fast: corrupt (non-UTF-8) bytes on a text type --- def test_invalid_utf8_fails_fast_typed() -> None: # Never a leaked UnicodeDecodeError — the "always typed" doctrine holds for # Door B too (cf. Door A wrapping path ValueError as SourceError). with pytest.raises(ExtractionError) as excinfo: extract_text("note.txt", b"\xffbad") assert excinfo.value.code == "extractor_decode_error" # --- fail-fast: empty CSV (no header row) --- def test_empty_csv_fails_fast() -> None: with pytest.raises(ExtractionError) as excinfo: extract_text("empty.csv", b"") assert excinfo.value.code == "extractor_empty_csv"