Chosen: a stdlib BMP reader, because `read_image` is on the CORE path and an asset's name is its content digest. Measured first, as the order requires: Pillow 12.3.0 IS in this tree (transitively under `pdfplumber`) and it DOES decode RLE8 correctly -- a hand-written stdlib decoder and Pillow agree on 19 of 19 of R761's real files, RGB per pixel. So the choice does not rest on capability. It rests on two properties of this package: `.html` and `.xml` carry images with no `[extract]` extra installed, so a Pillow converter either makes a core path depend on an optional binary wheel or buys the second runtime dependency; and encoding through an installed library would make a bundle's identity move with that library's version, which is the property 0.10.0 felled page rasterisation over and `encode_png`'s docstring already defends. Pillow keeps the job it is good for: the INDEPENDENT decoder in the tests, on neither side of the conversion. The defect, measured over the frozen R761 delivery's `assets/`, denominator 50: 29 JPEG, 2 PNG and 19 RLE8 BMP. The 19 are byte-correct files nothing reads, so 19 figures were present and invisible while `images: N` reported that they had arrived. - `VIEWABLE_MEDIA_TYPES` is tested against every asset's SNIFFED type, so it is a property and not a list of formats we met. WebP is on it and `sniff` does not recognise one; the limit is stated, not implied. - `bmp_to_png`: 8-bit uncompressed, 8-bit RLE8, 24-bit uncompressed. All five RLE8 opcodes. 19 of 19 real files convert with RGB identical to Pillow's decoding of the source, 2 366 365 pixels compared. - `asset_not_viewable` and `asset_bmp_unsupported`, both published, both leaving the concept's "not carried" line. - Traceability on the pointer's second line, where the rest of the asset metadata already lives: original media type, original sha256 in full, new sha256 in full. A converted asset is ONE asset. - The ceiling is paid on the DECLARATION before a row is allocated, and an RLE run is one clipped slice -- painting pixel by pixel leaves the memory bounded and the CPU unbounded. Two repairs the change forced, each measured rather than assumed: - `tests/test_assets.py`'s "dimensions absent is absent" used a TIFF, which is now refused before `read_image` returns. The property still has a reachable case -- a JPEG whose frame header never arrives -- and uses it. - `asset_holds` in the accounting gate proved a carry by hashing the SOURCE file, which a converted image's bundle cannot satisfy. It now also reads the two digests the bundle states and HASHES THE ASSET ITSELF, so a bundle claiming a conversion it did not perform still fails. `tools/okf_asset_census.py` is the committed instrument for the known-positive: one row per image, from two pinned trees. It was caught by the rule it serves -- its first version handed `_pdf_images` the wrong page object and reported 0 images over 67 PDFs with exit 0. The attribute is asserted now and a known-positive runs before the sweep. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
393 lines
16 KiB
Python
393 lines
16 KiB
Python
"""The published format promise, asserted rather than trusted.
|
|
|
|
`README.md` told consumers that `docx` and `xlsx` ship no parser and always
|
|
fail fast. That was true when it was written and became false the moment the
|
|
converter seam landed -- silently, because prose has no test.
|
|
|
|
This library already learned that lesson once: a published promise without a
|
|
test goes false without anyone noticing, and a guarantee made publicly is a
|
|
test obligation. So the README's claimed format list is compared against the
|
|
registries it describes. Adding a format without touching the README, or
|
|
describing one that does not exist, fails here.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from llm_ingestion_okf.extract import (
|
|
_CORE_EXTRACTORS,
|
|
_EVIDENCE,
|
|
_OPTIONAL_EXTRACTORS,
|
|
_PANDOC_FORMATS,
|
|
)
|
|
|
|
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
|
README = PROJECT_ROOT / "README.md"
|
|
|
|
# The line the README carries, and the one place this list is written in prose.
|
|
_FORMAT_LINE = re.compile(r"^<!-- extract-formats: (.+) -->$", re.MULTILINE)
|
|
|
|
|
|
def _declared_formats() -> set[str]:
|
|
match = _FORMAT_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README.md carries no `<!-- extract-formats: ... -->` marker; without "
|
|
"it this test cannot check the promise and the promise can drift"
|
|
)
|
|
return {token.strip() for token in match.group(1).split(",")}
|
|
|
|
|
|
def test_the_readme_names_exactly_the_formats_that_exist() -> None:
|
|
assert _declared_formats() == set(_CORE_EXTRACTORS) | set(_OPTIONAL_EXTRACTORS)
|
|
|
|
|
|
def test_the_readme_no_longer_claims_docx_and_xlsx_fail_fast() -> None:
|
|
"""The specific false sentence, pinned so it cannot come back.
|
|
|
|
Written as a search for the claim rather than for its exact wording: the
|
|
sentence could be rephrased and stay just as wrong.
|
|
"""
|
|
text = README.read_text(encoding="utf-8").lower()
|
|
for claim in (
|
|
"docx` and `xlsx` ship no parser",
|
|
"docx`/`xlsx` remain\nunimplemented",
|
|
"docx`/`xlsx` are still unimplemented",
|
|
):
|
|
assert claim.lower() not in text, f"README still claims: {claim}"
|
|
|
|
|
|
def test_the_readme_states_which_rows_are_not_measured() -> None:
|
|
"""A row that is not `measured` must not read as a supported one.
|
|
|
|
Three of the five office formats have denominator ZERO in the corpus this
|
|
work was measured on. A consumer reading the README should be able to see
|
|
that without reading the source.
|
|
|
|
Reads the CLASS from the table rather than the literal `unmeasured`: round
|
|
9 moved those three rows to `constructed`, and a test pinned to one word
|
|
would have gone green over an empty set the moment the word changed. Every
|
|
class that is not `measured` must be named in the README, whichever it is.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
weaker = {s.lstrip("."): e for s, e in _EVIDENCE.items() if e != "measured"}
|
|
assert weaker, "the evidence table lists no rows weaker than measured"
|
|
for suffix, evidence in weaker.items():
|
|
assert suffix in text, f"README does not mention the {evidence} row {suffix}"
|
|
assert evidence in text.lower(), f"README does not use the word {evidence}"
|
|
|
|
|
|
def test_the_readme_still_states_what_stays_out() -> None:
|
|
"""`.doc` (Word 97) and rastered PDFs are out, and stay named.
|
|
|
|
A format list that grows without also saying what it excludes reads as a
|
|
promise to handle anything office-shaped.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
assert ".doc`" in text or "Word 97" in text
|
|
assert ".doc" not in set(_PANDOC_FORMATS)
|
|
|
|
|
|
def test_the_readme_recursion_claim_matches_the_door() -> None:
|
|
"""The README says the drop directory is walked recursively. A sentence is
|
|
not a mechanism, so both halves are asserted here: the claim is in the
|
|
prose, and the door actually does it. Either one alone can go stale --
|
|
prose that outlived the code is the failure this whole module exists for.
|
|
"""
|
|
import tempfile
|
|
|
|
text = README.read_text(encoding="utf-8")
|
|
assert "walked **recursively**" in text
|
|
|
|
from llm_ingestion_okf.inbox import GateDecision, process_inbox
|
|
|
|
with tempfile.TemporaryDirectory() as workspace:
|
|
inbox = Path(workspace) / "inbox" / "sub"
|
|
inbox.mkdir(parents=True)
|
|
(inbox / "deep.md").write_text("Body\n", encoding="utf-8")
|
|
result = process_inbox(
|
|
Path(workspace) / "inbox",
|
|
Path(workspace) / "bundle",
|
|
"2026-09-07T08:00:00Z",
|
|
okf_type="reference",
|
|
gate=lambda body: GateDecision(sanitized_text=body, disposition="warn", reasons=()),
|
|
)
|
|
assert [item.source_file for item in result.persisted] == ["sub/deep.md"]
|
|
|
|
|
|
# --- the visible table, K3-26 ----------------------------------------------
|
|
#
|
|
# The comment marker above is machine-readable and invisible to a reader: the
|
|
# README's own prose named FIVE of the thirteen types the registry reads, and
|
|
# nothing went red, because the marker test only asks that the hidden list is
|
|
# complete. A reader does not read the marker. So the table a reader does see
|
|
# is pinned to the same registry, row for row, and to the evidence class the
|
|
# code records for each row.
|
|
|
|
_TABLE_HEADING = "## Supported file types"
|
|
|
|
# The evidence cell for a row `_EVIDENCE` does not carry. Those five are the
|
|
# stdlib rows: `_EVIDENCE` records a CORPUS class, and a row that has never
|
|
# been given one must not borrow `measured` from the row beside it.
|
|
_NO_CLASS = "stdlib, no corpus class"
|
|
|
|
|
|
def _table_rows() -> dict[str, list[str]]:
|
|
"""The table's data rows, keyed by suffix, with markup stripped per cell.
|
|
|
|
Backticks and asterisks are removed rather than matched, so the table can
|
|
be formatted freely and this test still reads what it says.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
assert _TABLE_HEADING in text, (
|
|
f"README.md carries no `{_TABLE_HEADING}` section; the format list is "
|
|
"then visible only in a hidden comment, which is what K3-26 fixed"
|
|
)
|
|
rows: dict[str, list[str]] = {}
|
|
for line in text.split(_TABLE_HEADING, 1)[1].splitlines():
|
|
stripped = line.strip()
|
|
if not stripped.startswith("|"):
|
|
if rows:
|
|
break
|
|
continue
|
|
cells = [re.sub(r"[`*]", "", cell).strip() for cell in stripped.strip("|").split("|")]
|
|
if cells and cells[0].startswith("."):
|
|
rows[cells[0]] = cells
|
|
return rows
|
|
|
|
|
|
def test_the_readme_table_names_every_type_the_registry_reads() -> None:
|
|
assert set(_table_rows()) == set(_CORE_EXTRACTORS) | set(_OPTIONAL_EXTRACTORS)
|
|
|
|
|
|
def test_the_readme_table_states_the_evidence_class_the_code_records() -> None:
|
|
rows = _table_rows()
|
|
assert rows, "no data rows found under the supported-file-types heading"
|
|
for suffix, cells in rows.items():
|
|
assert len(cells) >= 4, f"the {suffix} row has no evidence column: {cells}"
|
|
assert cells[3] == _EVIDENCE.get(suffix, _NO_CLASS), (
|
|
f"the {suffix} row says {cells[3]!r}; the code records "
|
|
f"{_EVIDENCE.get(suffix, _NO_CLASS)!r}"
|
|
)
|
|
|
|
|
|
def test_the_readme_table_separates_core_from_the_extract_extra() -> None:
|
|
"""Which rows need the optional extra is the first thing a consumer asks."""
|
|
for suffix, cells in _table_rows().items():
|
|
gated = "[extract]" in cells[2]
|
|
assert gated == (suffix in _OPTIONAL_EXTRACTORS), (
|
|
f"the {suffix} row's dependency cell reads {cells[2]!r}"
|
|
)
|
|
|
|
|
|
def test_the_readme_opening_does_not_name_five_of_thirteen() -> None:
|
|
"""The first thing a reader sees must not undersell what the code reads.
|
|
|
|
Either form passes: a pointer to the table, or the whole set spelled out.
|
|
A partial list -- the state before K3-26 -- passes neither.
|
|
"""
|
|
intro = README.read_text(encoding="utf-8").split("## Install", 1)[0]
|
|
if "#supported-file-types" in intro:
|
|
return
|
|
every = set(_CORE_EXTRACTORS) | set(_OPTIONAL_EXTRACTORS)
|
|
named = {s for s in every if re.search(rf"\b{s.lstrip('.')}\b", intro, re.IGNORECASE)}
|
|
assert named == every, f"the opening names {sorted(named)}, not all of {sorted(every)}"
|
|
|
|
|
|
def test_the_readme_carries_only_one_file_type_table() -> None:
|
|
"""A second table over the same rows is a copy nothing checks.
|
|
|
|
`### Binary extraction` carried its own six-row Format/Reader/Evidence
|
|
table until 2026-09-12. It was true when written, and it was reachable by
|
|
exactly the failure this module exists for: an evidence class copied into
|
|
prose that no test reads. Its rows now live in the pinned table alone.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
section = text.split("### Binary extraction", 1)[1].split("\n## ", 1)[0]
|
|
rows = [line for line in section.splitlines() if line.strip().startswith("|")]
|
|
assert not rows, f"a second file-type table is back under Binary extraction: {rows}"
|
|
|
|
|
|
# The `okf quality` thresholds, in the one place the README writes them. A bar
|
|
# published without a test goes false the way the format list did.
|
|
_THRESHOLD_LINE = re.compile(r"^<!-- quality-thresholds: (.+) -->$", re.MULTILINE)
|
|
|
|
THRESHOLD_DOCUMENT = PROJECT_ROOT / "docs" / "2026-09-12-g37-terskler.md"
|
|
|
|
|
|
def _declared_thresholds() -> dict[str, str]:
|
|
match = _THRESHOLD_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README.md carries no `<!-- quality-thresholds: ... -->` marker; without "
|
|
"it the published bars can drift from the ones the gate applies"
|
|
)
|
|
pairs = (token.strip().split("=") for token in match.group(1).split(","))
|
|
return {extension: share for extension, share in pairs}
|
|
|
|
|
|
def test_the_readme_names_exactly_the_thresholds_the_gate_applies() -> None:
|
|
from llm_ingestion_okf.quality import THRESHOLDS
|
|
|
|
assert _declared_thresholds() == {
|
|
extension: threshold.as_share() for extension, threshold in THRESHOLDS.items()
|
|
}
|
|
|
|
|
|
def test_the_threshold_document_carries_the_same_bars() -> None:
|
|
"""Three copies, one measurement: the code, the README and the document.
|
|
|
|
The document is where a bar's N and corpus live, so a bar that moved in the
|
|
code without moving there would publish a number nobody measured.
|
|
"""
|
|
from llm_ingestion_okf.quality import THRESHOLDS
|
|
|
|
text = THRESHOLD_DOCUMENT.read_text(encoding="utf-8")
|
|
for extension, threshold in THRESHOLDS.items():
|
|
row = f"| `{extension}` | `{threshold.metric}` | **{threshold.as_share()}** |"
|
|
assert row in text, f"{THRESHOLD_DOCUMENT.name} carries no row {row}"
|
|
|
|
|
|
def test_the_readme_quality_section_does_not_promise_a_quality_claim() -> None:
|
|
"""The one sentence that must not come back: PASS as a statement of quality."""
|
|
text = README.read_text(encoding="utf-8").lower()
|
|
assert "okf quality" in text
|
|
assert "regression bar against a pinned artifact" in text
|
|
|
|
|
|
# The `boundary_share` bar. Published in three places -- the code, the README
|
|
# and the threshold document -- and a bar published without a test goes false
|
|
# the way the format list did.
|
|
_BOUNDARY_LINE = re.compile(r"^<!-- quality-boundary-threshold: (.+) -->$", re.MULTILINE)
|
|
|
|
|
|
def test_the_readme_names_the_boundary_bar_the_gate_applies() -> None:
|
|
from llm_ingestion_okf.quality import BOUNDARY_THRESHOLD
|
|
|
|
match = _BOUNDARY_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README.md carries no `<!-- quality-boundary-threshold: ... -->` marker; "
|
|
"without it the published bar can drift from the one --fasit applies"
|
|
)
|
|
assert match.group(1).strip() == BOUNDARY_THRESHOLD.as_share()
|
|
|
|
|
|
def test_the_threshold_document_carries_the_boundary_bar_and_its_single_corpus() -> None:
|
|
"""N = 1 is half of what this bar is; a copy without it publishes the other half."""
|
|
from llm_ingestion_okf.quality import BOUNDARY_THRESHOLD
|
|
|
|
text = THRESHOLD_DOCUMENT.read_text(encoding="utf-8")
|
|
assert f"`{BOUNDARY_THRESHOLD.metric}` | **2 759/2 761**" in text
|
|
assert "1 corpus" in text
|
|
assert BOUNDARY_THRESHOLD.corpora == 1
|
|
|
|
|
|
# --- the CLI's default persist gate ----------------------------------------
|
|
#
|
|
# F1 (reported 2026-09-15): `okf build` injected a permissive stub and no
|
|
# argument in the package named a gate, so the command screened nothing while
|
|
# the guard was a mandatory runtime dependency and the README recommended a
|
|
# composition the command line could not reach. What made that survivable for
|
|
# months is that nothing tied the README's claim to the code's behaviour. This
|
|
# does. A published promise is a test obligation.
|
|
|
|
_GATE_LINE = re.compile(r"^<!-- cli-default-gate: (.+) -->$", re.MULTILINE)
|
|
|
|
|
|
def test_the_readme_names_the_gate_the_build_command_actually_defaults_to() -> None:
|
|
from llm_ingestion_okf import cli
|
|
|
|
match = _GATE_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README.md carries no `<!-- cli-default-gate: ... -->` marker; without it "
|
|
"the documented default can drift from the one the command applies, which "
|
|
"is exactly how F1 survived"
|
|
)
|
|
assert match.group(1).strip() == cli.DEFAULT_GATE
|
|
|
|
|
|
def test_the_readme_names_every_gate_the_command_accepts() -> None:
|
|
"""A tier reachable but undocumented is a tier nobody can choose.
|
|
|
|
The stricter one matters most: a caller whose folder IS an untrusted drop
|
|
has to be able to find `guard-user-upload` without reading the source.
|
|
"""
|
|
from llm_ingestion_okf.corpus import GATE_NAMES
|
|
|
|
text = README.read_text(encoding="utf-8")
|
|
for name in GATE_NAMES:
|
|
assert f"`{name}`" in text, f"README does not name the gate {name}"
|
|
|
|
|
|
# --- the asset default is published and pinned (0.10.0) --------------------
|
|
#
|
|
# Same obligation as the gate marker above it, for the same reason: 0.10.0
|
|
# changes what `okf build` writes for every consumer whose sources carry
|
|
# pictures, and a documented default that can drift from the applied one is how
|
|
# F1 survived for months.
|
|
|
|
_ASSETS_LINE = re.compile(r"^<!-- cli-default-assets: (on|off) -->$", re.MULTILINE)
|
|
|
|
|
|
def test_the_readme_names_the_asset_default_the_build_command_applies() -> None:
|
|
from llm_ingestion_okf import cli
|
|
|
|
match = _ASSETS_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README.md carries no `<!-- cli-default-assets: ... -->` marker; without it "
|
|
"the documented default can drift from the one the command applies"
|
|
)
|
|
assert (match.group(1) == "on") is cli.DEFAULT_ASSETS
|
|
|
|
|
|
def test_the_readme_names_the_opt_out_that_reproduces_the_old_bytes() -> None:
|
|
text = README.read_text(encoding="utf-8")
|
|
assert "`--no-assets`" in text
|
|
assert "NOT CARRIED" in text
|
|
|
|
|
|
def test_the_readme_states_that_image_bytes_are_not_screened() -> None:
|
|
"""The boundary, published rather than left to be discovered.
|
|
|
|
The guard is text-only. A consumer weighing an untrusted drop has to be
|
|
able to learn which half of a concept was looked at without reading this
|
|
package's source.
|
|
"""
|
|
text = README.read_text(encoding="utf-8")
|
|
assert "image bytes are not screened" in text.lower()
|
|
|
|
|
|
_VIEWABLE_LINE = re.compile(
|
|
r"^<!-- asset-viewable-media-types: ([a-z0-9/,+.-]+) -->$", re.MULTILINE
|
|
)
|
|
|
|
|
|
def test_the_readme_publishes_the_viewable_set_the_code_applies() -> None:
|
|
"""A published set is a test obligation, the same as a published bound.
|
|
|
|
Sorted on both sides so the marker states a SET and not an order, and
|
|
compared as a whole rather than by membership: a README naming three of
|
|
four would pass every containment check and still tell a consumer that a
|
|
format is refused when it is carried.
|
|
"""
|
|
from llm_ingestion_okf.assets import VIEWABLE_MEDIA_TYPES
|
|
|
|
match = _VIEWABLE_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, (
|
|
"README carries no `<!-- asset-viewable-media-types: ... -->` marker; without it "
|
|
"the set a consumer reads and the set the code applies can drift apart silently"
|
|
)
|
|
assert match.group(1).split(",") == sorted(VIEWABLE_MEDIA_TYPES)
|
|
|
|
|
|
_MAX_PIXELS_LINE = re.compile(r"^<!-- asset-max-pixels: (\d+) -->$", re.MULTILINE)
|
|
|
|
|
|
def test_the_readme_publishes_the_pixel_bound_the_code_applies() -> None:
|
|
"""A published bound is a test obligation: the number in the README is the
|
|
number that refuses an image."""
|
|
from llm_ingestion_okf.assets import MAX_IMAGE_PIXELS
|
|
|
|
match = _MAX_PIXELS_LINE.search(README.read_text(encoding="utf-8"))
|
|
assert match is not None, "README carries no `<!-- asset-max-pixels: N -->` marker"
|
|
assert int(match.group(1)) == MAX_IMAGE_PIXELS
|