"""Door B extraction registry: file bytes -> text, per file type (Phase 2 step 1). Core stdlib extractors (md/txt/csv/json/html) plus the fail-fast gates: unknown extension, the [extract]-gated binary types when the extra is absent, corrupt (non-UTF-8) bytes, and an empty CSV. All file-type->text extraction lives HERE (the guard is text-only); this step adds zero runtime dependency and no guard call — it is pure, deterministic plumbing. """ from __future__ import annotations import importlib.util import sys from pathlib import Path import pytest from llm_ingestion_okf import ExtractionError, ExtractionWarning, extract_text FIXTURES = Path(__file__).parent / "fixtures" # The parser-dependent tests below need the optional extra, so they are skipped # without it — an optional extra that forced its parser on every test run would # not be optional. The behaviour that must survive EITHER WAY is the rejection, # and that one is asserted unconditionally above via the import probe. requires_extract = pytest.mark.skipif( importlib.util.find_spec("pdfplumber") is None, reason="the optional [extract] extra is not installed", ) # --- core extractors: happy path (a fixture per type) --- def test_md_passthrough() -> None: assert extract_text("note.md", b"# Title\n\nBody\n") == "# Title\n\nBody\n" def test_txt_passthrough() -> None: assert extract_text("note.txt", b"plain text") == "plain text" def test_utf8_sig_bom_is_stripped() -> None: # utf-8-sig so a BOM never leaks into the first character (baseline parity # with Door A's read_csv). assert extract_text("note.txt", b"\xef\xbb\xbfhello") == "hello" def test_csv_renders_the_phase_1_markdown_table() -> None: out = extract_text("data.csv", b"name,age\nAda,36\n") assert out == "| name | age |\n| --- | --- |\n| Ada | 36 |\n" def test_json_is_verbatim_inside_a_fenced_block() -> None: out = extract_text("cfg.json", b'{"k": 1}\n') assert out == '```\n{"k": 1}\n```\n' def test_html_text_via_htmlparser() -> None: # Block boundaries separate words; tags themselves contribute no text. out = extract_text("page.html", b"

Title

Hello world

") assert out == "Title Hello world" def test_html_skips_script_and_style() -> None: html = b"

Keep

" assert extract_text("page.html", html) == "Keep" def test_htm_is_an_html_alias() -> None: assert extract_text("page.htm", b"

hi

") == "hi" def test_extension_dispatch_is_case_insensitive() -> None: assert extract_text("NOTE.MD", b"x") == "x" # --- fail-fast: unknown extension --- def test_unknown_extension_fails_fast() -> None: with pytest.raises(ExtractionError) as excinfo: extract_text("archive.zip", b"PK\x03\x04") assert excinfo.value.code == "extractor_unknown" def test_missing_extension_fails_fast() -> None: with pytest.raises(ExtractionError) as excinfo: extract_text("README", b"x") assert excinfo.value.code == "extractor_unknown" # --- fail-fast: an [extract]-gated type without the extra installed --- def test_optional_type_without_the_extra_fails_fast( monkeypatch: pytest.MonkeyPatch, ) -> None: """`extractor_extra_missing` reached through the import probe, not a suffix set. This test used to reach the code through a `.docx` or `.xlsx` filename, which sat in `_UNPARSED_OPTIONAL_EXTENSIONS` because the extra shipped no parser for those types. That set is on its way to empty: once those types gain a converter, a membership test can no longer raise this code at all, and a test pinned to it would go red for the right reason at the worst moment. The import probe is the durable path — it is how the gate actually works (`extract.py`'s `_extract_pdf`), and it stays reachable no matter how many types gain parsers. """ monkeypatch.setitem(sys.modules, "pdfplumber", None) with pytest.raises(ExtractionError) as excinfo: extract_text("report.pdf", b"%PDF-1.4") assert excinfo.value.code == "extractor_extra_missing" # The error names the extra so the operator knows the remedy — never a # silent skip, never a bundled parser in core. assert "extract" in str(excinfo.value) def test_pdf_without_the_extra_installed_keeps_the_same_rejection( monkeypatch: pytest.MonkeyPatch, ) -> None: """The gate became an import probe; the rejection did not change. Setting the module to None in sys.modules is what CPython treats as a failed import, so this exercises the uninstalled path REGARDLESS of whether the extra is installed in the running environment — a skip would have preserved nothing on the machine where the parser is present. """ monkeypatch.setitem(sys.modules, "pdfplumber", None) with pytest.raises(ExtractionError) as excinfo: extract_text("doc.pdf", b"%PDF-1.4") assert excinfo.value.code == "extractor_extra_missing" assert "extract" in str(excinfo.value) # --- office types: the table-driven converter seam --- def test_every_office_row_names_its_reader() -> None: """The table is the contract; these are all the rows there are. Written as an exact equality rather than a series of `in` checks: a row added without a fixture and without an evidence class is the failure mode this pins, and only equality catches an ADDITION. """ from llm_ingestion_okf.extract import _PANDOC_FORMATS assert _PANDOC_FORMATS == { ".docx": "docx", ".xlsx": "xlsx", ".pptx": "pptx", ".odt": "odt", ".rtf": "rtf", } def test_html_and_epub_are_excluded_from_the_converter_on_purpose() -> None: """`.html` has a stdlib extractor; routing it through the converter would buy nothing and would add CVE-2025-51591 (SSRF via an iframe in HTML input), which is unpatched in every converter version. `.epub` is out for the same "no gain" half of that reason. """ from llm_ingestion_okf.extract import _CORE_EXTRACTORS, _PANDOC_FORMATS assert ".html" not in _PANDOC_FORMATS assert ".epub" not in _PANDOC_FORMATS assert ".html" in _CORE_EXTRACTORS def test_evidence_class_is_asserted_not_commented() -> None: """Which rows were measured is a fact about this work, not a footnote. `.pptx`, `.odt` and `.rtf` have denominator ZERO in the corpus this arm was measured on. A comment saying so rots; an assertion that names them keeps an unmeasured row from quietly presenting as a supported one. """ from llm_ingestion_okf.extract import _EVIDENCE, _PANDOC_FORMATS assert set(_EVIDENCE) == set(_PANDOC_FORMATS), "every row needs an evidence class" assert {s for s, e in _EVIDENCE.items() if e == "measured"} == {".docx", ".xlsx"} assert {s for s, e in _EVIDENCE.items() if e == "unmeasured"} == { ".pptx", ".odt", ".rtf", } def test_the_unparsed_set_is_empty_now_that_every_row_has_a_reader() -> None: """The membership branch that used to raise `extractor_extra_missing` has no members left. This is why both tests for that code were repointed at the import probe before this step landed. """ from llm_ingestion_okf.extract import _UNPARSED_OPTIONAL_EXTENSIONS assert _UNPARSED_OPTIONAL_EXTENSIONS == frozenset() def test_the_converter_is_called_with_the_load_bearing_arguments( monkeypatch: pytest.MonkeyPatch, ) -> None: """`--eol=lf --wrap=none` and the markdown writer, all three measured. Not hygiene, and not style. The defaults produce DIFFERENT BYTES (maximum line length 75 against 447), which a byte-pinned golden would register as a change nobody made. And `-t plain` destroys the headings the segment proposer reads: measured, a document yielding 15 entries including two real headings yields 13 with none once the writer is `plain`. """ import llm_ingestion_okf.extract as extract_module seen: dict[str, object] = {} def fake_convert(source: object, to: str, format: str, extra_args: object) -> str: seen.update(to=to, format=format, extra_args=list(extra_args)) # type: ignore[call-overload] return "# Heading\n\nBody." monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert) with pytest.warns(ExtractionWarning): text = extract_text("note.docx", b"PK\x03\x04") assert text == "# Heading\n\nBody." assert seen["to"] == "markdown" assert seen["format"] == "docx" assert "--eol=lf" in seen["extra_args"] # type: ignore[operator] assert "--wrap=none" in seen["extra_args"] # type: ignore[operator] @pytest.mark.parametrize( ("name", "reader"), [ ("a.docx", "docx"), ("b.xlsx", "xlsx"), ("c.pptx", "pptx"), ("d.odt", "odt"), ("e.rtf", "rtf"), ], ) def test_each_row_dispatches_to_its_reader( name: str, reader: str, monkeypatch: pytest.MonkeyPatch ) -> None: import llm_ingestion_okf.extract as extract_module seen: dict[str, object] = {} def fake_convert(source: object, to: str, format: str, extra_args: object) -> str: seen["format"] = format return "text" monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert) with pytest.warns(ExtractionWarning): extract_text(name, b"bytes") assert seen["format"] == reader def test_an_empty_conversion_is_refused_not_persisted( monkeypatch: pytest.MonkeyPatch, ) -> None: """Same reason as `extractor_empty_pdf`: an empty concept is the silent skip this registry exists to prevent.""" import llm_ingestion_okf.extract as extract_module monkeypatch.setattr( extract_module, "_convert_bytes", lambda source, to, format, extra_args: " \n\t ", ) with pytest.raises(ExtractionError) as excinfo: extract_text("empty.docx", b"bytes") assert excinfo.value.code == "extractor_empty_conversion" def test_a_converter_failure_is_wrapped_never_leaked( monkeypatch: pytest.MonkeyPatch, ) -> None: import llm_ingestion_okf.extract as extract_module def boom(source: object, to: str, format: str, extra_args: object) -> str: raise RuntimeError("pandoc said no") monkeypatch.setattr(extract_module, "_convert_bytes", boom) with pytest.raises(ExtractionError) as excinfo: extract_text("bad.docx", b"bytes") assert excinfo.value.code == "extractor_convert_error" assert "pandoc said no" in str(excinfo.value) def test_office_conversion_warns_that_it_is_lossy( monkeypatch: pytest.MonkeyPatch, ) -> None: """Drawn content does not survive extraction here either, and the warning is emitted after the parse -- a run that produced no text has nothing to be lossy about.""" import llm_ingestion_okf.extract as extract_module monkeypatch.setattr( extract_module, "_convert_bytes", lambda source, to, format, extra_args: "body", ) with pytest.warns(ExtractionWarning): extract_text("note.docx", b"bytes") # --- pdf: the [extract] parser (pdfplumber) --- # The expected text is frozen against a committed fixture ON PURPOSE. Extraction # is deterministic within a parser version and NOT guaranteed across one # (pdfplumber pins pdfminer.six==20260107 exactly; pdfminer.six ships # date-stamped releases with no stability contract). This literal is what makes # a parser upgrade break something visible instead of drifting silently. KRAV_TEXT = "Krav til helning på utkilingen\n60 og 70 1:15" @requires_extract def test_pdf_extracts_text_with_label_and_value_on_one_line() -> None: data = (FIXTURES / "two-line-krav.pdf").read_bytes() with pytest.warns(ExtractionWarning): assert extract_text("krav.pdf", data) == KRAV_TEXT @requires_extract def test_pdf_extraction_is_byte_stable_across_calls() -> None: """Determinism is a promise this library already makes; hold it here too.""" data = (FIXTURES / "two-line-krav.pdf").read_bytes() with pytest.warns(ExtractionWarning): first = extract_text("krav.pdf", data) second = extract_text("krav.pdf", data) assert first == second @requires_extract def test_pdf_extraction_warns_that_drawn_content_is_not_recovered() -> None: """Figures are vector drawings: only the caption survives leg 2. Categorically true of text extraction, so it is stated as a warning on every PDF rather than guessed at per document — detecting "is there a figure here" would be exactly the layout heuristic this order declined. """ data = (FIXTURES / "two-line-krav.pdf").read_bytes() with pytest.warns(ExtractionWarning, match="figures"): extract_text("krav.pdf", data) @requires_extract def test_pdf_with_no_text_layer_fails_fast() -> None: """A scanned PDF yields nothing; an empty concept would be a silent skip.""" data = (FIXTURES / "no-text-layer.pdf").read_bytes() with pytest.raises(ExtractionError) as excinfo: extract_text("scan.pdf", data) assert excinfo.value.code == "extractor_empty_pdf" @requires_extract def test_corrupt_pdf_fails_fast_typed() -> None: """Never a leaked pdfminer exception — the "always typed" doctrine holds.""" with pytest.raises(ExtractionError) as excinfo: extract_text("broken.pdf", b"not a pdf at all") assert excinfo.value.code == "extractor_pdf_error" # --- fail-fast: corrupt (non-UTF-8) bytes on a text type --- def test_invalid_utf8_fails_fast_typed() -> None: # Never a leaked UnicodeDecodeError — the "always typed" doctrine holds for # Door B too (cf. Door A wrapping path ValueError as SourceError). with pytest.raises(ExtractionError) as excinfo: extract_text("note.txt", b"\xffbad") assert excinfo.value.code == "extractor_decode_error" # --- fail-fast: empty CSV (no header row) --- def test_empty_csv_fails_fast() -> None: with pytest.raises(ExtractionError) as excinfo: extract_text("empty.csv", b"") assert excinfo.value.code == "extractor_empty_csv"