"""Regenerate the committed PDF fixtures for tests/test_extract.py. Hand-written minimal PDFs: objects laid out by hand, xref offsets computed from the emitted bytes. No generator library, so the fixtures are auditable byte for byte and reproducible from this file alone. Run from the repository root: python3 tests/fixtures/make_fixtures.py """ from __future__ import annotations from pathlib import Path HERE = Path(__file__).parent # Two text lines: a heading, and one requirement row with label and value on # the SAME line. That pairing is the property the parser choice was made on # (see docs/2026-08-21-g2-pdf-extraction-measurement.md), so the fixture # fails visibly if a parser upgrade ever breaks it. Byte 0xE5 is the Norwegian # 'a-ring' in WinAnsiEncoding, which the font object below declares. KRAV_CONTENT = ( b"BT /F1 12 Tf 20 160 Td (Krav til helning p\xe5 utkilingen) Tj ET\n" b"BT /F1 12 Tf 20 140 Td (60 og 70 1:15) Tj ET\n" ) # A structurally valid page carrying no text operators at all -- the shape a # scanned or image-only PDF presents to a text extractor. NO_TEXT_CONTENT = b"20 20 160 160 re S\n" def build_pdf(content: bytes) -> bytes: """Assemble a one-page PDF around `content` as the page content stream.""" objects = [ b"<< /Type /Catalog /Pages 2 0 R >>", b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] " b"/Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>", b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"endstream", b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>", ] out = bytearray(b"%PDF-1.4\n") offsets = [] for number, body in enumerate(objects, start=1): offsets.append(len(out)) out += str(number).encode() + b" 0 obj\n" + body + b"\nendobj\n" xref_at = len(out) size = str(len(objects) + 1).encode() out += b"xref\n0 " + size + b"\n0000000000 65535 f \n" for offset in offsets: out += ("%010d 00000 n \n" % offset).encode() out += b"trailer\n<< /Size " + size + b" /Root 1 0 R >>\n" out += b"startxref\n" + str(xref_at).encode() + b"\n%%EOF\n" return bytes(out) if __name__ == "__main__": for name, content in ( ("two-line-krav.pdf", KRAV_CONTENT), ("no-text-layer.pdf", NO_TEXT_CONTENT), ): (HERE / name).write_bytes(build_pdf(content)) print(f"wrote {name}")