Round 9: the four rests in STATE's NESTE that needed no operator decision.
CLAUSE 1 CLASSIFIED BY THE NUMBER, NOT THE TITLE. `_TRAILING_PAGE_NUMBER`
admitted a candidate into a contents run by asking whether the title ended in
an integer -- a question about the number. A drawing's dimension chain, a
schematic's labels, a door schedule, a coordinate column and a soil-layer
table all end in integers and name nothing. Measured over the 43-document
corpus: 68 candidates discarded over 11 of 39 readable documents, of which
19 over 5 documents are data rows.
That corrects round 8's own decomposition. Its "four misclassified numeric
tables and seven real contents listings" needs each document on one side, and
two of the eleven are both. Read across all 68 titles rather than the
three-title sample: 5 documents carry a data row, 8 carry a real entry.
`--contents-name` requires a NAME to survive stripping the page number. The
threshold is SWEPT, not chosen, and collapses at both ends: at an alphabetic
run of 1 a door schedule keeps a stray `V` and 13 of 19 are rescued; at 3 the
two-letter section name `VA` stops being a name, falls out of run membership,
and takes `RIB`, `MMI` and `Tittelfelt` below `CONTENTS_RUN` with it -- one
acronym costing four REAL entries. At 2: 16 of 19 rescued, 0 of 49 regressed.
The three not rescued carry a real word and are named rather than rounded off.
THE CONVERTER'S ANCHOR WAS IN THE CONCEPT ID. Pandoc writes a sheet as
`## <name> {#sheet-N}` and a titled slide as `## <title> {#slide-N}`. Because
a filename is reduced FROM the title, the anchor reached both. Operator
authorised the strip 2026-09-09 after the exposure was counted: 2 of 810
concepts on the previous default bundle, 2 of 1108 on Arm B, 1 of 26 on the
operator's folder. Two ids renamed, one of which `portfolio-optimiser` has
cited in writing; both are in the report so that message can be sent.
One rule in one function, read by BOTH title-forming sites -- a rule in only
one would leave the id and the title naming the same concept differently. The
known-negative is the point: `Mal for {kundenavn}` is a title an author wrote.
odt/rtf/pptx MEASURED END TO END FOR THE FIRST TIME, on hand-built documents,
because the corpus denominator is genuinely zero (86 files: 66 pdf, 10 docx,
4 xlsx, 2 zip, 2 smc, 2 doc). `_EVIDENCE` gains a third class rather than
stretching an existing one: `constructed` means the row has met a document,
but not one anyone wrote for their own purposes. odt 1 of 1 declared headings;
pptx 2 of 2 on a deck that declares slide titles and 0 of 2 on one that does
not -- round 7's reading of pptx was a fixture property, not the format; rtf
0 segments, because the container has no heading style and the author's title
is bold text. rtf is the one open finding.
ACCEPTANCE, all four. The 12-position reference is label-identical in BOTH
readings (pdf 7/8, docx 3/3, xlsx 0/1 or 1/1, sheet 10/12 or 11/12). One K2
bundle carrying both changes: 453 concepts / 865 md, hit@8 [1,1,1,1,1,None]
on it AND on Arm B, with the known-negative still reproducing on the new
bytes. `okf project` byte-equal to `okf build`, `diff -r` empty. Consumer
cost is a re-run: 436/832 -> 453/865, digest 21af4a1aa98315cf.
Three published numbers corrected: README's 596 tests (1515), README's "15
concepts out" for `okf project` (that was the O6 defect; it is 26), and O6's
print-mode method, which does not reproduce without --allowedTools.
Report: docs/2026-09-09-k3-runde9-restene.md
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
631 lines
24 KiB
Python
631 lines
24 KiB
Python
"""Door B extraction registry: file bytes -> text, per file type (Phase 2 step 1).
|
|
|
|
Core stdlib extractors (md/txt/csv/json/html) plus the fail-fast gates: unknown
|
|
extension, the [extract]-gated binary types when the extra is absent, corrupt
|
|
(non-UTF-8) bytes, and an empty CSV. All file-type->text extraction lives HERE
|
|
(the guard is text-only); this step adds zero runtime dependency and no guard
|
|
call — it is pure, deterministic plumbing.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import importlib.util
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf import ExtractionError, ExtractionWarning, extract_text
|
|
|
|
FIXTURES = Path(__file__).parent / "fixtures"
|
|
|
|
# The parser-dependent tests below need the optional extra, so they are skipped
|
|
# without it — an optional extra that forced its parser on every test run would
|
|
# not be optional. The behaviour that must survive EITHER WAY is the rejection,
|
|
# and that one is asserted unconditionally above via the import probe.
|
|
requires_extract = pytest.mark.skipif(
|
|
importlib.util.find_spec("pdfplumber") is None,
|
|
reason="the optional [extract] extra is not installed",
|
|
)
|
|
|
|
# --- core extractors: happy path (a fixture per type) ---
|
|
|
|
|
|
def test_md_passthrough() -> None:
|
|
assert extract_text("note.md", b"# Title\n\nBody\n") == "# Title\n\nBody\n"
|
|
|
|
|
|
def test_txt_passthrough() -> None:
|
|
assert extract_text("note.txt", b"plain text") == "plain text"
|
|
|
|
|
|
def test_utf8_sig_bom_is_stripped() -> None:
|
|
# utf-8-sig so a BOM never leaks into the first character (baseline parity
|
|
# with Door A's read_csv).
|
|
assert extract_text("note.txt", b"\xef\xbb\xbfhello") == "hello"
|
|
|
|
|
|
def test_csv_renders_the_phase_1_markdown_table() -> None:
|
|
out = extract_text("data.csv", b"name,age\nAda,36\n")
|
|
assert out == "| name | age |\n| --- | --- |\n| Ada | 36 |\n"
|
|
|
|
|
|
def test_json_is_verbatim_inside_a_fenced_block() -> None:
|
|
out = extract_text("cfg.json", b'{"k": 1}\n')
|
|
assert out == '```\n{"k": 1}\n```\n'
|
|
|
|
|
|
def test_html_text_via_htmlparser() -> None:
|
|
# Block boundaries separate words; tags themselves contribute no text.
|
|
out = extract_text("page.html", b"<h1>Title</h1><p>Hello <b>world</b></p>")
|
|
assert out == "Title Hello world"
|
|
|
|
|
|
def test_html_skips_script_and_style() -> None:
|
|
html = b"<style>.x{color:red}</style><p>Keep</p><script>evil()</script>"
|
|
assert extract_text("page.html", html) == "Keep"
|
|
|
|
|
|
def test_htm_is_an_html_alias() -> None:
|
|
assert extract_text("page.htm", b"<p>hi</p>") == "hi"
|
|
|
|
|
|
def test_extension_dispatch_is_case_insensitive() -> None:
|
|
assert extract_text("NOTE.MD", b"x") == "x"
|
|
|
|
|
|
# --- fail-fast: unknown extension ---
|
|
|
|
|
|
def test_unknown_extension_fails_fast() -> None:
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("archive.zip", b"PK\x03\x04")
|
|
assert excinfo.value.code == "extractor_unknown"
|
|
|
|
|
|
def test_missing_extension_fails_fast() -> None:
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("README", b"x")
|
|
assert excinfo.value.code == "extractor_unknown"
|
|
|
|
|
|
# --- fail-fast: an [extract]-gated type without the extra installed ---
|
|
|
|
|
|
def test_optional_type_without_the_extra_fails_fast(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""`extractor_extra_missing` reached through the import probe, not a suffix set.
|
|
|
|
This test used to reach the code through a `.docx` or `.xlsx` filename,
|
|
which sat in `_UNPARSED_OPTIONAL_EXTENSIONS` because the extra shipped no
|
|
parser for those types. That set is on its way to empty: once those types gain a converter,
|
|
a membership test can no longer raise this code at all, and a test pinned
|
|
to it would go red for the right reason at the worst moment.
|
|
|
|
The import probe is the durable path — it is how the gate actually works
|
|
(`extract.py`'s `_extract_pdf`), and it stays reachable no matter how many
|
|
types gain parsers.
|
|
"""
|
|
monkeypatch.setitem(sys.modules, "pdfplumber", None)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("report.pdf", b"%PDF-1.4")
|
|
assert excinfo.value.code == "extractor_extra_missing"
|
|
# The error names the extra so the operator knows the remedy — never a
|
|
# silent skip, never a bundled parser in core.
|
|
assert "extract" in str(excinfo.value)
|
|
|
|
|
|
def test_pdf_without_the_extra_installed_keeps_the_same_rejection(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""The gate became an import probe; the rejection did not change.
|
|
|
|
Setting the module to None in sys.modules is what CPython treats as a
|
|
failed import, so this exercises the uninstalled path REGARDLESS of
|
|
whether the extra is installed in the running environment — a skip would
|
|
have preserved nothing on the machine where the parser is present.
|
|
"""
|
|
monkeypatch.setitem(sys.modules, "pdfplumber", None)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("doc.pdf", b"%PDF-1.4")
|
|
assert excinfo.value.code == "extractor_extra_missing"
|
|
assert "extract" in str(excinfo.value)
|
|
|
|
|
|
# --- office types: the table-driven converter seam ---
|
|
|
|
|
|
def test_every_office_row_names_its_reader() -> None:
|
|
"""The table is the contract; these are all the rows there are.
|
|
|
|
Written as an exact equality rather than a series of `in` checks: a row
|
|
added without a fixture and without an evidence class is the failure mode
|
|
this pins, and only equality catches an ADDITION.
|
|
"""
|
|
from llm_ingestion_okf.extract import _PANDOC_FORMATS
|
|
|
|
assert _PANDOC_FORMATS == {
|
|
".docx": "docx",
|
|
".xlsx": "xlsx",
|
|
".pptx": "pptx",
|
|
".odt": "odt",
|
|
".rtf": "rtf",
|
|
}
|
|
|
|
|
|
def test_html_and_epub_are_excluded_from_the_converter_on_purpose() -> None:
|
|
"""`.html` has a stdlib extractor; routing it through the converter would
|
|
buy nothing and would add CVE-2025-51591 (SSRF via an iframe in HTML
|
|
input), which is unpatched in every converter version. `.epub` is out for
|
|
the same "no gain" half of that reason.
|
|
"""
|
|
from llm_ingestion_okf.extract import _CORE_EXTRACTORS, _PANDOC_FORMATS
|
|
|
|
assert ".html" not in _PANDOC_FORMATS
|
|
assert ".epub" not in _PANDOC_FORMATS
|
|
assert ".html" in _CORE_EXTRACTORS
|
|
|
|
|
|
def test_evidence_class_is_asserted_not_commented() -> None:
|
|
"""Which rows were measured is a fact about this work, not a footnote.
|
|
|
|
`.pptx`, `.odt` and `.rtf` have denominator ZERO in the corpus this arm was
|
|
measured on. A comment saying so rots; an assertion that names them keeps
|
|
an unmeasured row from quietly presenting as a supported one.
|
|
"""
|
|
from llm_ingestion_okf.extract import _EVIDENCE, _PANDOC_FORMATS
|
|
|
|
assert set(_EVIDENCE) == set(_PANDOC_FORMATS), "every row needs an evidence class"
|
|
assert {s for s, e in _EVIDENCE.items() if e == "measured"} == {".docx", ".xlsx"}
|
|
# Since 2026-09-09 the three office rows are `constructed`, not
|
|
# `unmeasured`: each has now been put through end to end on a hand-built
|
|
# document with a hand-written fasit, and none of them has a corpus file.
|
|
# The set is asserted EMPTY rather than dropped -- a class with no members
|
|
# is a fact about this package, and a future row can re-enter it.
|
|
assert {s for s, e in _EVIDENCE.items() if e == "constructed"} == {
|
|
".pptx",
|
|
".odt",
|
|
".rtf",
|
|
}
|
|
assert {s for s, e in _EVIDENCE.items() if e == "unmeasured"} == set()
|
|
|
|
|
|
def test_the_unparsed_set_is_empty_now_that_every_row_has_a_reader() -> None:
|
|
"""The membership branch that used to raise `extractor_extra_missing` has
|
|
no members left. This is why both tests for that code were repointed at the
|
|
import probe before this step landed.
|
|
"""
|
|
from llm_ingestion_okf.extract import _UNPARSED_OPTIONAL_EXTENSIONS
|
|
|
|
assert _UNPARSED_OPTIONAL_EXTENSIONS == frozenset()
|
|
|
|
|
|
def test_the_converter_is_called_with_the_load_bearing_arguments(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""`--eol=lf --wrap=none` and the markdown writer, all three measured.
|
|
|
|
Not hygiene, and not style. The defaults produce DIFFERENT BYTES (maximum
|
|
line length 75 against 447), which a byte-pinned golden would register as a
|
|
change nobody made. And `-t plain` destroys the headings the segment
|
|
proposer reads: measured, a document yielding 15 entries including two real
|
|
headings yields 13 with none once the writer is `plain`.
|
|
"""
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
seen: dict[str, object] = {}
|
|
|
|
def fake_convert(source: object, to: str, format: str, extra_args: object) -> str:
|
|
seen.update(to=to, format=format, extra_args=list(extra_args)) # type: ignore[call-overload]
|
|
return "# Heading\n\nBody."
|
|
|
|
monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert)
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text("note.docx", b"PK\x03\x04")
|
|
|
|
assert text == "# Heading\n\nBody."
|
|
assert seen["to"] == "markdown"
|
|
assert seen["format"] == "docx"
|
|
assert "--eol=lf" in seen["extra_args"] # type: ignore[operator]
|
|
assert "--wrap=none" in seen["extra_args"] # type: ignore[operator]
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("name", "reader"),
|
|
[
|
|
("a.docx", "docx"),
|
|
("b.xlsx", "xlsx"),
|
|
("c.pptx", "pptx"),
|
|
("d.odt", "odt"),
|
|
("e.rtf", "rtf"),
|
|
],
|
|
)
|
|
def test_each_row_dispatches_to_its_reader(
|
|
name: str, reader: str, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
seen: dict[str, object] = {}
|
|
|
|
def fake_convert(source: object, to: str, format: str, extra_args: object) -> str:
|
|
seen["format"] = format
|
|
return "text"
|
|
|
|
monkeypatch.setattr(extract_module, "_convert_bytes", fake_convert)
|
|
with pytest.warns(ExtractionWarning):
|
|
extract_text(name, b"bytes")
|
|
assert seen["format"] == reader
|
|
|
|
|
|
def test_an_empty_conversion_is_refused_not_persisted(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""Same reason as `extractor_empty_pdf`: an empty concept is the silent
|
|
skip this registry exists to prevent."""
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
monkeypatch.setattr(
|
|
extract_module,
|
|
"_convert_bytes",
|
|
lambda source, to, format, extra_args: " \n\t ",
|
|
)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("empty.docx", b"bytes")
|
|
assert excinfo.value.code == "extractor_empty_conversion"
|
|
|
|
|
|
def test_a_converter_failure_is_wrapped_never_leaked(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
def boom(source: object, to: str, format: str, extra_args: object) -> str:
|
|
raise RuntimeError("pandoc said no")
|
|
|
|
monkeypatch.setattr(extract_module, "_convert_bytes", boom)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("bad.docx", b"bytes")
|
|
assert excinfo.value.code == "extractor_convert_error"
|
|
assert "pandoc said no" in str(excinfo.value)
|
|
|
|
|
|
def test_office_conversion_warns_that_it_is_lossy(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""Drawn content does not survive extraction here either, and the warning
|
|
is emitted after the parse -- a run that produced no text has nothing to be
|
|
lossy about."""
|
|
import llm_ingestion_okf.extract as extract_module
|
|
|
|
monkeypatch.setattr(
|
|
extract_module,
|
|
"_convert_bytes",
|
|
lambda source, to, format, extra_args: "body",
|
|
)
|
|
with pytest.warns(ExtractionWarning):
|
|
extract_text("note.docx", b"bytes")
|
|
|
|
|
|
# --- pdf: the [extract] parser (pdfplumber) ---
|
|
|
|
# The expected text is frozen against a committed fixture ON PURPOSE. Extraction
|
|
# is deterministic within a parser version and NOT guaranteed across one
|
|
# (pdfplumber pins pdfminer.six==20260107 exactly; pdfminer.six ships
|
|
# date-stamped releases with no stability contract). This literal is what makes
|
|
# a parser upgrade break something visible instead of drifting silently.
|
|
KRAV_TEXT = "Krav til helning på utkilingen\n60 og 70 1:15"
|
|
|
|
# The office fixtures are frozen the same way, and against a NAMED converter
|
|
# version -- a frozen literal means nothing without one, because the thing it
|
|
# pins is "this converter, on this input, produces these bytes". `_pandoc.py`
|
|
# refuses any other version, so the two pins hold each other up.
|
|
requires_pandoc = pytest.mark.skipif(
|
|
importlib.util.find_spec("pypandoc") is None,
|
|
reason="the optional [extract] extra is not installed",
|
|
)
|
|
|
|
# docx: a heading and one requirement row with label and value on the SAME
|
|
# line, mirroring the property the PDF fixture pins.
|
|
DOCX_TEXT = "# Krav til helning\n\n60 og 70 1:15"
|
|
|
|
# xlsx: the sheet name becomes a heading and the rows become a table. The
|
|
# label/value pairing survives on one row, which is the property that matters.
|
|
#
|
|
# THIS LITERAL MOVED ONCE, deliberately, and the move is the fix reported in
|
|
# `docs/2026-09-08-prisform-og-loggen-k2.md`: the spreadsheet row now writes
|
|
# pipe tables, so the cells arrive delimited instead of padded. Every character
|
|
# of content is the same; only the table form changed.
|
|
XLSX_TEXT = "## Krav {#sheet-1}\n\n| Krav til helning | |\n|----|----|\n| 60 og 70 | 1:15 |"
|
|
|
|
# The negative control, committed rather than described: the SAME document
|
|
# without `word/styles.xml`. The body survives and the heading marker does not.
|
|
DOCX_NO_STYLES_TEXT = "Krav til helning\n\n60 og 70 1:15"
|
|
|
|
|
|
@requires_pandoc
|
|
def test_docx_extracts_to_its_frozen_text() -> None:
|
|
data = (FIXTURES / "two-line-krav.docx").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
assert extract_text("krav.docx", data) == DOCX_TEXT
|
|
|
|
|
|
@requires_pandoc
|
|
def test_xlsx_extracts_to_its_frozen_text() -> None:
|
|
data = (FIXTURES / "two-line-krav.xlsx").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
assert extract_text("krav.xlsx", data) == XLSX_TEXT
|
|
|
|
|
|
@requires_pandoc
|
|
def test_a_docx_without_a_styles_part_loses_its_heading() -> None:
|
|
"""The negative control for the fixture policy, run rather than asserted.
|
|
|
|
`word/styles.xml` is what makes the converter see a heading. Without it the
|
|
same document extracts as flat prose -- so a fixture built WITHOUT that
|
|
part would pin the body and pin nothing at all about structure, while
|
|
looking exactly as convincing.
|
|
|
|
Structure is the half the segment proposer reads, which is why this is a
|
|
committed fixture and not a sentence in a README.
|
|
"""
|
|
data = (FIXTURES / "no-styles-krav.docx").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text("krav.docx", data)
|
|
assert text == DOCX_NO_STYLES_TEXT
|
|
assert not text.startswith("#"), "the heading marker must be absent"
|
|
assert DOCX_TEXT.startswith("#"), "and present in the fixture that has styles"
|
|
|
|
|
|
@requires_pandoc
|
|
def test_the_frozen_office_text_is_pinned_to_a_named_converter_version() -> None:
|
|
"""A frozen literal without a named version pins nothing.
|
|
|
|
If the converter version ever moves, these literals must be re-measured
|
|
rather than trusted -- so the version is asserted right where they live.
|
|
"""
|
|
from llm_ingestion_okf._pandoc import PANDOC_VERSION, resolve_pandoc
|
|
|
|
assert PANDOC_VERSION == "3.9"
|
|
assert resolve_pandoc().is_file()
|
|
|
|
|
|
@requires_pandoc
|
|
def test_office_extraction_is_byte_stable_across_calls() -> None:
|
|
data = (FIXTURES / "two-line-krav.docx").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
first = extract_text("krav.docx", data)
|
|
with pytest.warns(ExtractionWarning):
|
|
second = extract_text("krav.docx", data)
|
|
assert first == second
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_extracts_text_with_label_and_value_on_one_line() -> None:
|
|
data = (FIXTURES / "two-line-krav.pdf").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
assert extract_text("krav.pdf", data) == KRAV_TEXT
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_extraction_is_byte_stable_across_calls() -> None:
|
|
"""Determinism is a promise this library already makes; hold it here too."""
|
|
data = (FIXTURES / "two-line-krav.pdf").read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
first = extract_text("krav.pdf", data)
|
|
second = extract_text("krav.pdf", data)
|
|
assert first == second
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_extraction_warns_that_drawn_content_is_not_recovered() -> None:
|
|
"""Figures are vector drawings: only the caption survives leg 2.
|
|
|
|
Categorically true of text extraction, so it is stated as a warning on
|
|
every PDF rather than guessed at per document — detecting "is there a
|
|
figure here" would be exactly the layout heuristic this order declined.
|
|
"""
|
|
data = (FIXTURES / "two-line-krav.pdf").read_bytes()
|
|
with pytest.warns(ExtractionWarning, match="figures"):
|
|
extract_text("krav.pdf", data)
|
|
|
|
|
|
@requires_extract
|
|
def test_pdf_with_no_text_layer_fails_fast() -> None:
|
|
"""A scanned PDF yields nothing; an empty concept would be a silent skip."""
|
|
data = (FIXTURES / "no-text-layer.pdf").read_bytes()
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("scan.pdf", data)
|
|
assert excinfo.value.code == "extractor_empty_pdf"
|
|
|
|
|
|
@requires_extract
|
|
def test_corrupt_pdf_fails_fast_typed() -> None:
|
|
"""Never a leaked pdfminer exception — the "always typed" doctrine holds."""
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("broken.pdf", b"not a pdf at all")
|
|
assert excinfo.value.code == "extractor_pdf_error"
|
|
|
|
|
|
# --- fail-fast: corrupt (non-UTF-8) bytes on a text type ---
|
|
|
|
|
|
def test_invalid_utf8_fails_fast_typed() -> None:
|
|
# Never a leaked UnicodeDecodeError — the "always typed" doctrine holds for
|
|
# Door B too (cf. Door A wrapping path ValueError as SourceError).
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("note.txt", b"\xffbad")
|
|
assert excinfo.value.code == "extractor_decode_error"
|
|
|
|
|
|
# --- fail-fast: empty CSV (no header row) ---
|
|
|
|
|
|
def test_empty_csv_fails_fast() -> None:
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
extract_text("empty.csv", b"")
|
|
assert excinfo.value.code == "extractor_empty_csv"
|
|
|
|
|
|
# --- xlsx: the spreadsheet form a consumer has to read ---
|
|
#
|
|
# Measured on K2 and reported upstream by the first consumer to read the bundle
|
|
# with a live model (`docs/2026-09-08-prisform-og-loggen-k2.md`): the converter's
|
|
# DEFAULT markdown writer emits simple tables, which pad every cell out to the
|
|
# width of the widest cell in its column. One long prose cell therefore turns
|
|
# every other row in that column into a whitespace carpet -- 887 characters
|
|
# between a label and its amount on the real sheet -- while the header row names
|
|
# a single column because only the first source cell in row 1 is filled. The
|
|
# bytes reach the reader and the STRUCTURE does not.
|
|
#
|
|
# `prisark.xlsx` is that shape in miniature, hand-laid rather than recorded, and
|
|
# it carries its own negative control on a second sheet.
|
|
|
|
PRISARK = "prisark.xlsx"
|
|
|
|
|
|
def _table_rows(text: str) -> list[list[str]]:
|
|
"""Every pipe-table row in `text`, as its cells, in order."""
|
|
rows = []
|
|
for line in text.split("\n"):
|
|
stripped = line.strip()
|
|
if not (stripped.startswith("|") and stripped.endswith("|")):
|
|
continue
|
|
cells = [cell.strip() for cell in stripped[1:-1].split("|")]
|
|
if all(set(cell) <= set("-:") and cell for cell in cells):
|
|
continue # the header separator is punctuation, not a row
|
|
rows.append(cells)
|
|
return rows
|
|
|
|
|
|
@requires_pandoc
|
|
def test_a_spreadsheet_keeps_its_columns_one_row_per_line() -> None:
|
|
"""The label and the amount arrive as separate cells on one line.
|
|
|
|
This is the property the whole change exists for. Asserted as properties
|
|
rather than only as a frozen literal, because a literal pins bytes and says
|
|
nothing about which of them was the point.
|
|
"""
|
|
data = (FIXTURES / PRISARK).read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text(PRISARK, data)
|
|
|
|
rows = _table_rows(text)
|
|
assert [
|
|
"01",
|
|
"Rigging og drift av byggeplass, medregnet alt som ikke er "
|
|
"priset spesifikt nedenfor og alt som er innkalkulert i de angitte "
|
|
"prisene",
|
|
"5647500",
|
|
] in rows
|
|
assert ["02", "Andel", "12.5"] in rows
|
|
|
|
longest = max((len(run) for run in re.findall(r" {2,}", text)), default=0)
|
|
assert longest <= 8, f"a whitespace run of {longest} is a carpet, not a column"
|
|
|
|
|
|
@requires_pandoc
|
|
def test_an_integral_amount_loses_the_converters_decimal_and_a_real_one_keeps_it() -> None:
|
|
"""`5647500` is a number; `92.0` in the same sheet is TEXT.
|
|
|
|
The converter renders both as `<digits>.0`, so the output alone cannot tell
|
|
them apart. The shared string table can, and is what the rewrite consults --
|
|
which is why this test asserts both directions from ONE document.
|
|
"""
|
|
data = (FIXTURES / PRISARK).read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text(PRISARK, data)
|
|
|
|
assert "5647500.0" not in text
|
|
assert "250000.0" not in text
|
|
assert "12.5" in text, "a genuine decimal is a value, not a converter artefact"
|
|
assert ["03", "92.0", "250000"] in _table_rows(text), (
|
|
"a shared-string cell reading 92.0 is author text and survives verbatim"
|
|
)
|
|
assert "Kode 4 \\| 5.0" in text, (
|
|
"a `5.0` INSIDE a cell is not a cell: the delimiter test is what sees that"
|
|
)
|
|
|
|
|
|
@requires_pandoc
|
|
def test_a_single_column_sheet_gains_no_columns() -> None:
|
|
"""The negative control, in the same document as the case it controls.
|
|
|
|
Sheet 2 has ONE column in the source. There is nothing to recover, so the
|
|
fix must not invent a second cell anywhere on it. Its three values arrive
|
|
in order and alone.
|
|
|
|
It is NOT byte-identical before and after the change, and that is measured
|
|
rather than glossed: the writer emits a pipe table for every table it
|
|
writes, so a one-column table changes delimiter form too. What must not
|
|
change is the cell content and the column count.
|
|
"""
|
|
data = (FIXTURES / PRISARK).read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text(PRISARK, data)
|
|
|
|
single = text.split("## Enkeltkolonne")[1]
|
|
rows = _table_rows(single)
|
|
assert [cells for cells in rows if any(cells)] == [
|
|
["Notat"],
|
|
["Ingen kolonner her"],
|
|
["Sum ikke oppgitt"],
|
|
]
|
|
|
|
|
|
# The scoping control. The same writer change applied to the other four office
|
|
# rows was MEASURED to move them (the odt fixture 1366 -> 1105 characters), so
|
|
# this digest can fail; it is not a tautology. The change is deliberately
|
|
# spreadsheet-only: a spreadsheet IS a grid and has no prose fallback, while
|
|
# moving docx/pptx/odt/rtf would move a corpus denominator nothing has measured.
|
|
# A red here means the writer stopped being scoped -- read the diff and decide.
|
|
OFFICE_TEXT_DIGESTS = {
|
|
"k2-office/krav-tekstdokument.odt": (
|
|
"58c9776f0d7f2b2a3a9d2774e4ae243b265c31b5b6b96914ef4db419fa66e4e2"
|
|
),
|
|
"k2-office/krav-presentasjon.pptx": (
|
|
"752420a04d651a416938ff9f0b3c2de5849bd2ea1d52063dd88aaae65ab99b90"
|
|
),
|
|
"k2-office/krav-rikt-tekstformat.rtf": (
|
|
"79cbf756eb482bb603f82c171d11efe74b9ba62ab9ef679ecd6bb8c3b3740ffb"
|
|
),
|
|
}
|
|
|
|
|
|
@requires_pandoc
|
|
@pytest.mark.parametrize("relative", sorted(OFFICE_TEXT_DIGESTS))
|
|
def test_the_other_office_rows_are_untouched_by_the_spreadsheet_writer(relative: str) -> None:
|
|
path = FIXTURES / relative
|
|
with pytest.warns(ExtractionWarning):
|
|
text = extract_text(path.name, path.read_bytes())
|
|
assert hashlib.sha256(text.encode("utf-8")).hexdigest() == OFFICE_TEXT_DIGESTS[relative]
|
|
|
|
|
|
# The whole fixture, frozen against the same named converter version as the
|
|
# literals above. The property tests say WHAT matters; this one catches any
|
|
# other byte moving without anybody noticing.
|
|
PRISARK_TEXT = (
|
|
"## Prisark {#sheet-1}\n\n"
|
|
"| Prisskjema | | |\n"
|
|
"|----|----|----|\n"
|
|
"| Post | Beskrivelse | Sum |\n"
|
|
"| 01 | Rigging og drift av byggeplass, medregnet alt som ikke er priset "
|
|
"spesifikt nedenfor og alt som er innkalkulert i de angitte prisene | 5647500 |\n"
|
|
"| 02 | Andel | 12.5 |\n"
|
|
"| 03 | 92.0 | 250000 |\n"
|
|
"| 04 | Kode 4 \\| 5.0 | |\n\n"
|
|
"## Enkeltkolonne {#sheet-2}\n\n"
|
|
"| Notat |\n"
|
|
"|----|\n"
|
|
"| Ingen kolonner her |\n"
|
|
"| Sum ikke oppgitt |"
|
|
)
|
|
|
|
|
|
@requires_pandoc
|
|
def test_prisark_extracts_to_its_frozen_text() -> None:
|
|
data = (FIXTURES / PRISARK).read_bytes()
|
|
with pytest.warns(ExtractionWarning):
|
|
assert extract_text(PRISARK, data) == PRISARK_TEXT
|