"""Regenerate the committed fixtures for tests/test_extract.py.
Hand-written minimal documents: PDF objects laid out by hand with xref offsets
computed from the emitted bytes, and OOXML containers assembled part by part.
No generator library anywhere, so every fixture is auditable byte for byte and
reproducible from this file alone.
THE POLICY IS WHAT FORBIDS THE SHORTCUT. A `.docx` written by the converter and
then read by the converter proves only that the converter agrees with itself --
it would stay green through any conversion defect that is symmetric, which is
most of them. Hand-laying the parts is what makes the fixture an independent
statement about the format rather than a recording of our own output.
Run from the repository root: python3 tests/fixtures/make_fixtures.py
"""
from __future__ import annotations
import io
import zipfile
from pathlib import Path
HERE = Path(__file__).parent
# Two text lines: a heading, and one requirement row with label and value on
# the SAME line. That pairing is the property the parser choice was made on
# (see docs/2026-08-21-g2-pdf-extraction-measurement.md), so the fixture
# fails visibly if a parser upgrade ever breaks it. Byte 0xE5 is the Norwegian
# 'a-ring' in WinAnsiEncoding, which the font object below declares.
KRAV_CONTENT = (
b"BT /F1 12 Tf 20 160 Td (Krav til helning p\xe5 utkilingen) Tj ET\n"
b"BT /F1 12 Tf 20 140 Td (60 og 70 1:15) Tj ET\n"
)
# A structurally valid page carrying no text operators at all -- the shape a
# scanned or image-only PDF presents to a text extractor.
NO_TEXT_CONTENT = b"20 20 160 160 re S\n"
def build_pdf(content: bytes) -> bytes:
"""Assemble a one-page PDF around `content` as the page content stream."""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] "
b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>",
b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"endstream",
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>",
]
out = bytearray(b"%PDF-1.4\n")
offsets = []
for number, body in enumerate(objects, start=1):
offsets.append(len(out))
out += str(number).encode() + b" 0 obj\n" + body + b"\nendobj\n"
xref_at = len(out)
size = str(len(objects) + 1).encode()
out += b"xref\n0 " + size + b"\n0000000000 65535 f \n"
for offset in offsets:
out += ("%010d 00000 n \n" % offset).encode()
out += b"trailer\n<< /Size " + size + b" /Root 1 0 R >>\n"
out += b"startxref\n" + str(xref_at).encode() + b"\n%%EOF\n"
return bytes(out)
# --- office containers -------------------------------------------------------
#
# A fixed timestamp on every member, because a zip records mtime and the whole
# point is a byte-reproducible file: without it the fixture would differ on
# every regeneration and `git diff --quiet` could never be the check.
_ZIP_DATE = (2020, 1, 1, 0, 0, 0)
_XML = ''
# `word/styles.xml` IS REQUIRED, not decoration. Measured during planning: the
# same document WITHOUT a styles part extracts as plain text with no heading
# marker at all, so a fixture lacking it would pin the body and silently pin
# nothing about structure -- which is the half the segment proposer reads.
_DOCX_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _XML
+ ''
+ ''
+ "",
"word/_rels/document.xml.rels": _XML
+ ''
+ ''
+ "",
"word/styles.xml": _XML
+ ''
+ ''
+ "",
"word/document.xml": _XML
+ ''
+ 'Krav til helning'
+ "60 og 70 1:15"
+ "",
}
# The same document MINUS the styles part. A negative control, committed rather
# than described: it is what proves the styles part is load-bearing, and a
# claim of that kind that nothing runs is a claim that decays.
_DOCX_NO_STYLES_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _DOCX_PARTS["_rels/.rels"],
"word/document.xml": _DOCX_PARTS["word/document.xml"],
}
# A minimal SpreadsheetML workbook: one sheet, a heading row and a label/value
# row, mirroring what the PDF fixture does for its format.
#
# THE SHARED STRING TABLE IS NOT A STYLE CHOICE. The first attempt used inline
# strings (`t="inlineStr"`), which is valid SpreadsheetML and which the
# converter reads as EMPTY CELLS -- the sheet name survived and every value
# vanished, with exit code 0 and no warning. A `dimension` element and a shared
# string table are what make the values arrive. This is the same class of
# defect as the missing `styles.xml`: structurally valid input, silently
# reduced output, nothing anywhere saying so.
_XLSX_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _XML
+ ''
+ ''
+ "",
"xl/_rels/workbook.xml.rels": _XML
+ ''
+ ''
+ ''
+ "",
"xl/workbook.xml": _XML
+ ''
+ '',
"xl/sharedStrings.xml": _XML
+ ''
+ "Krav til helning60 og 701:15",
"xl/worksheets/sheet1.xml": _XML
+ ''
+ ''
+ '0
'
+ '12
'
+ "",
}
def build_ooxml(parts: dict[str, str]) -> bytes:
"""Zip the parts with a fixed timestamp and no compression variance.
A constant `date_time` on every member is what makes the output
reproducible: a zip records mtime, so the default `ZipFile.writestr` would
stamp the current time and the fixture would differ on every run --
which would make `git diff --quiet` useless as the regeneration check.
"""
out = io.BytesIO()
with zipfile.ZipFile(out, "w", compression=zipfile.ZIP_DEFLATED) as archive:
for name, payload in parts.items():
info = zipfile.ZipInfo(name, date_time=_ZIP_DATE)
info.compress_type = zipfile.ZIP_DEFLATED
archive.writestr(info, payload)
return out.getvalue()
if __name__ == "__main__":
for name, content in (
("two-line-krav.pdf", KRAV_CONTENT),
("no-text-layer.pdf", NO_TEXT_CONTENT),
):
(HERE / name).write_bytes(build_pdf(content))
print(f"wrote {name}")
for name, parts in (
("two-line-krav.docx", _DOCX_PARTS),
("no-styles-krav.docx", _DOCX_NO_STYLES_PARTS),
("two-line-krav.xlsx", _XLSX_PARTS),
):
(HERE / name).write_bytes(build_ooxml(parts))
print(f"wrote {name}")