test(extract): reach extra-missing through the import probe
Both tests reached `extractor_extra_missing` through a `.docx`/`.xlsx` filename, which works only while `_UNPARSED_OPTIONAL_EXTENSIONS` is non-empty. Those types are about to gain a converter, which empties the set and makes the membership branch unreachable -- the tests would have gone red for the right reason at the worst moment, mid-series. Repointed both at the import probe, the mechanism the pdf gate already uses and the one path that stays reachable however many types gain parsers. Measured negative control: without the probe the same call raises `extractor_pdf_error`, so the probe is load-bearing and the test can still fail. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
11cde6d0b5
commit
5f524f3902
2 changed files with 26 additions and 9 deletions
|
|
@ -12,6 +12,7 @@ import hashlib
|
|||
import importlib.util
|
||||
import json
|
||||
import sqlite3
|
||||
import sys
|
||||
import urllib.error
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
|
@ -344,12 +345,14 @@ def test_extractor_unknown() -> None:
|
|||
assert code_of(excinfo) == "extractor_unknown"
|
||||
|
||||
|
||||
def test_extractor_extra_missing() -> None:
|
||||
# `docx` rather than `pdf`: the extra now ships a pdf parser, so the type
|
||||
# that still has none is what proves this code is reachable. The pdf
|
||||
# import-probe path is covered in tests/test_extract.py.
|
||||
def test_extractor_extra_missing(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# The import probe rather than a suffix: `docx` proved this code reachable
|
||||
# only while the extra shipped no parser for it. Once it gains a converter,
|
||||
# `_UNPARSED_OPTIONAL_EXTENSIONS` empties and the membership branch becomes
|
||||
# unreachable — the probe is the path that survives.
|
||||
monkeypatch.setitem(sys.modules, "pdfplumber", None)
|
||||
with pytest.raises(ExtractionError) as excinfo:
|
||||
extract_text("doc.docx", b"binary")
|
||||
extract_text("report.pdf", b"%PDF-1.4")
|
||||
assert code_of(excinfo) == "extractor_extra_missing"
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -89,13 +89,27 @@ def test_missing_extension_fails_fast() -> None:
|
|||
assert excinfo.value.code == "extractor_unknown"
|
||||
|
||||
|
||||
# --- fail-fast: [extract]-gated binary types the extra ships no parser for ---
|
||||
# --- fail-fast: an [extract]-gated type without the extra installed ---
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", ["doc.docx", "sheet.xlsx"])
|
||||
def test_optional_type_without_a_parser_fails_fast(name: str) -> None:
|
||||
def test_optional_type_without_the_extra_fails_fast(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""`extractor_extra_missing` reached through the import probe, not a suffix set.
|
||||
|
||||
This test used to reach the code through a `.docx` or `.xlsx` filename,
|
||||
which sat in `_UNPARSED_OPTIONAL_EXTENSIONS` because the extra shipped no
|
||||
parser for those types. That set is on its way to empty: once those types gain a converter,
|
||||
a membership test can no longer raise this code at all, and a test pinned
|
||||
to it would go red for the right reason at the worst moment.
|
||||
|
||||
The import probe is the durable path — it is how the gate actually works
|
||||
(`extract.py`'s `_extract_pdf`), and it stays reachable no matter how many
|
||||
types gain parsers.
|
||||
"""
|
||||
monkeypatch.setitem(sys.modules, "pdfplumber", None)
|
||||
with pytest.raises(ExtractionError) as excinfo:
|
||||
extract_text(name, b"binary")
|
||||
extract_text("report.pdf", b"%PDF-1.4")
|
||||
assert excinfo.value.code == "extractor_extra_missing"
|
||||
# The error names the extra so the operator knows the remedy — never a
|
||||
# silent skip, never a bundled parser in core.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue