The extractor registry reads 13 extensions. The README's opening line named five of them, and the full list existed only in a hidden `<!-- extract-formats: ... -->` comment, which no reader reads -- so the README undersold what the code does and stated no evidence class anywhere a consumer would look. A `## Supported file types` table now carries one row per extension: reader, dependency (core or the `[extract]` extra), the evidence class `_EVIDENCE` records for the row, and one honest note. The three `constructed` office rows carry their denominators (N = 1, N = 2, N = 1) in the table itself, so a row that has met no document anyone wrote cannot read as a supported one; `.htm` does not borrow `.html`'s 828-file class, because the code records none for it. A `Not read today` section states the absences (`.doc`, `.epub`, `.eml`/`.msg`, image files, source files, `.one`/`.vsd`) as facts, not as a queue. Test first, red before the table existed: four assertions in `tests/test_docs_promises.py` pin the table's row set to `_CORE_EXTRACTORS | _OPTIONAL_EXTRACTORS`, each evidence cell to `_EVIDENCE` (and to a fixed `stdlib, no corpus class` where the code records none), the core/extra split to the registries, and the opening to the table. No change to `extract.py` and no version bump: nothing about what is read moved, only what the README says about it. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
195 lines
8.1 KiB
Python
195 lines
8.1 KiB
Python
"""The published format promise, asserted rather than trusted.
|
|
|
|
`README.md` told consumers that `docx` and `xlsx` ship no parser and always
|
|
fail fast. That was true when it was written and became false the moment the
|
|
converter seam landed -- silently, because prose has no test.
|
|
|
|
This library already learned that lesson once: a published promise without a
|
|
test goes false without anyone noticing, and a guarantee made publicly is a
|
|
test obligation. So the README's claimed format list is compared against the
|
|
registries it describes. Adding a format without touching the README, or
|
|
describing one that does not exist, fails here.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from llm_ingestion_okf.extract import (
|
|
_CORE_EXTRACTORS,
|
|
_EVIDENCE,
|
|
_OPTIONAL_EXTRACTORS,
|
|
_PANDOC_FORMATS,
|
|
)
|
|
|
|
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
|
README = PROJECT_ROOT / "README.md"
|
|
|
|
# The line the README carries, and the one place this list is written in prose.
|
|
_FORMAT_LINE = re.compile(r"^<!-- extract-formats: (.+) -->$", re.MULTILINE)
|
|
|
|
|
|
def _declared_formats() -> set[str]:
|
|
match = _FORMAT_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README.md carries no `<!-- extract-formats: ... -->` marker; without "
|
|
"it this test cannot check the promise and the promise can drift"
|
|
)
|
|
return {token.strip() for token in match.group(1).split(",")}
|
|
|
|
|
|
def test_the_readme_names_exactly_the_formats_that_exist() -> None:
|
|
assert _declared_formats() == set(_CORE_EXTRACTORS) | set(_OPTIONAL_EXTRACTORS)
|
|
|
|
|
|
def test_the_readme_no_longer_claims_docx_and_xlsx_fail_fast() -> None:
|
|
"""The specific false sentence, pinned so it cannot come back.
|
|
|
|
Written as a search for the claim rather than for its exact wording: the
|
|
sentence could be rephrased and stay just as wrong.
|
|
"""
|
|
text = README.read_text(encoding="utf-8").lower()
|
|
for claim in (
|
|
"docx` and `xlsx` ship no parser",
|
|
"docx`/`xlsx` remain\nunimplemented",
|
|
"docx`/`xlsx` are still unimplemented",
|
|
):
|
|
assert claim.lower() not in text, f"README still claims: {claim}"
|
|
|
|
|
|
def test_the_readme_states_which_rows_are_not_measured() -> None:
|
|
"""A row that is not `measured` must not read as a supported one.
|
|
|
|
Three of the five office formats have denominator ZERO in the corpus this
|
|
work was measured on. A consumer reading the README should be able to see
|
|
that without reading the source.
|
|
|
|
Reads the CLASS from the table rather than the literal `unmeasured`: round
|
|
9 moved those three rows to `constructed`, and a test pinned to one word
|
|
would have gone green over an empty set the moment the word changed. Every
|
|
class that is not `measured` must be named in the README, whichever it is.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
weaker = {s.lstrip("."): e for s, e in _EVIDENCE.items() if e != "measured"}
|
|
assert weaker, "the evidence table lists no rows weaker than measured"
|
|
for suffix, evidence in weaker.items():
|
|
assert suffix in text, f"README does not mention the {evidence} row {suffix}"
|
|
assert evidence in text.lower(), f"README does not use the word {evidence}"
|
|
|
|
|
|
def test_the_readme_still_states_what_stays_out() -> None:
|
|
"""`.doc` (Word 97) and rastered PDFs are out, and stay named.
|
|
|
|
A format list that grows without also saying what it excludes reads as a
|
|
promise to handle anything office-shaped.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
assert ".doc`" in text or "Word 97" in text
|
|
assert ".doc" not in set(_PANDOC_FORMATS)
|
|
|
|
|
|
def test_the_readme_recursion_claim_matches_the_door() -> None:
|
|
"""The README says the drop directory is walked recursively. A sentence is
|
|
not a mechanism, so both halves are asserted here: the claim is in the
|
|
prose, and the door actually does it. Either one alone can go stale --
|
|
prose that outlived the code is the failure this whole module exists for.
|
|
"""
|
|
import tempfile
|
|
|
|
text = README.read_text(encoding="utf-8")
|
|
assert "walked **recursively**" in text
|
|
|
|
from llm_ingestion_okf.inbox import GateDecision, process_inbox
|
|
|
|
with tempfile.TemporaryDirectory() as workspace:
|
|
inbox = Path(workspace) / "inbox" / "sub"
|
|
inbox.mkdir(parents=True)
|
|
(inbox / "deep.md").write_text("Body\n", encoding="utf-8")
|
|
result = process_inbox(
|
|
Path(workspace) / "inbox",
|
|
Path(workspace) / "bundle",
|
|
"2026-09-07T08:00:00Z",
|
|
okf_type="reference",
|
|
gate=lambda body: GateDecision(sanitized_text=body, disposition="warn", reasons=()),
|
|
)
|
|
assert [item.source_file for item in result.persisted] == ["sub/deep.md"]
|
|
|
|
|
|
# --- the visible table, K3-26 ----------------------------------------------
|
|
#
|
|
# The comment marker above is machine-readable and invisible to a reader: the
|
|
# README's own prose named FIVE of the thirteen types the registry reads, and
|
|
# nothing went red, because the marker test only asks that the hidden list is
|
|
# complete. A reader does not read the marker. So the table a reader does see
|
|
# is pinned to the same registry, row for row, and to the evidence class the
|
|
# code records for each row.
|
|
|
|
_TABLE_HEADING = "## Supported file types"
|
|
|
|
# The evidence cell for a row `_EVIDENCE` does not carry. Those five are the
|
|
# stdlib rows: `_EVIDENCE` records a CORPUS class, and a row that has never
|
|
# been given one must not borrow `measured` from the row beside it.
|
|
_NO_CLASS = "stdlib, no corpus class"
|
|
|
|
|
|
def _table_rows() -> dict[str, list[str]]:
|
|
"""The table's data rows, keyed by suffix, with markup stripped per cell.
|
|
|
|
Backticks and asterisks are removed rather than matched, so the table can
|
|
be formatted freely and this test still reads what it says.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
assert _TABLE_HEADING in text, (
|
|
f"README.md carries no `{_TABLE_HEADING}` section; the format list is "
|
|
"then visible only in a hidden comment, which is what K3-26 fixed"
|
|
)
|
|
rows: dict[str, list[str]] = {}
|
|
for line in text.split(_TABLE_HEADING, 1)[1].splitlines():
|
|
stripped = line.strip()
|
|
if not stripped.startswith("|"):
|
|
if rows:
|
|
break
|
|
continue
|
|
cells = [re.sub(r"[`*]", "", cell).strip() for cell in stripped.strip("|").split("|")]
|
|
if cells and cells[0].startswith("."):
|
|
rows[cells[0]] = cells
|
|
return rows
|
|
|
|
|
|
def test_the_readme_table_names_every_type_the_registry_reads() -> None:
|
|
assert set(_table_rows()) == set(_CORE_EXTRACTORS) | set(_OPTIONAL_EXTRACTORS)
|
|
|
|
|
|
def test_the_readme_table_states_the_evidence_class_the_code_records() -> None:
|
|
rows = _table_rows()
|
|
assert rows, "no data rows found under the supported-file-types heading"
|
|
for suffix, cells in rows.items():
|
|
assert len(cells) >= 4, f"the {suffix} row has no evidence column: {cells}"
|
|
assert cells[3] == _EVIDENCE.get(suffix, _NO_CLASS), (
|
|
f"the {suffix} row says {cells[3]!r}; the code records "
|
|
f"{_EVIDENCE.get(suffix, _NO_CLASS)!r}"
|
|
)
|
|
|
|
|
|
def test_the_readme_table_separates_core_from_the_extract_extra() -> None:
|
|
"""Which rows need the optional extra is the first thing a consumer asks."""
|
|
for suffix, cells in _table_rows().items():
|
|
gated = "[extract]" in cells[2]
|
|
assert gated == (suffix in _OPTIONAL_EXTRACTORS), (
|
|
f"the {suffix} row's dependency cell reads {cells[2]!r}"
|
|
)
|
|
|
|
|
|
def test_the_readme_opening_does_not_name_five_of_thirteen() -> None:
|
|
"""The first thing a reader sees must not undersell what the code reads.
|
|
|
|
Either form passes: a pointer to the table, or the whole set spelled out.
|
|
A partial list -- the state before K3-26 -- passes neither.
|
|
"""
|
|
intro = README.read_text(encoding="utf-8").split("## Install", 1)[0]
|
|
if "#supported-file-types" in intro:
|
|
return
|
|
every = set(_CORE_EXTRACTORS) | set(_OPTIONAL_EXTRACTORS)
|
|
named = {s for s in every if re.search(rf"\b{s.lstrip('.')}\b", intro, re.IGNORECASE)}
|
|
assert named == every, f"the opening names {sorted(named)}, not all of {sorted(every)}"
|