"""Regenerate the committed fixtures for tests/test_extract.py.
Hand-written minimal documents: PDF objects laid out by hand with xref offsets
computed from the emitted bytes, and OOXML containers assembled part by part.
No generator library anywhere, so every fixture is auditable byte for byte and
reproducible from this file alone.
THE POLICY IS WHAT FORBIDS THE SHORTCUT. A `.docx` written by the converter and
then read by the converter proves only that the converter agrees with itself --
it would stay green through any conversion defect that is symmetric, which is
most of them. Hand-laying the parts is what makes the fixture an independent
statement about the format rather than a recording of our own output.
Run from the repository root: python3 tests/fixtures/make_fixtures.py
"""
from __future__ import annotations
import io
import zipfile
from pathlib import Path
HERE = Path(__file__).parent
# Two text lines: a heading, and one requirement row with label and value on
# the SAME line. That pairing is the property the parser choice was made on
# (see docs/2026-08-21-g2-pdf-extraction-measurement.md), so the fixture
# fails visibly if a parser upgrade ever breaks it. Byte 0xE5 is the Norwegian
# 'a-ring' in WinAnsiEncoding, which the font object below declares.
KRAV_CONTENT = (
b"BT /F1 12 Tf 20 160 Td (Krav til helning p\xe5 utkilingen) Tj ET\n"
b"BT /F1 12 Tf 20 140 Td (60 og 70 1:15) Tj ET\n"
)
# A structurally valid page carrying no text operators at all -- the shape a
# scanned or image-only PDF presents to a text extractor.
NO_TEXT_CONTENT = b"20 20 160 160 re S\n"
# One line of text per page, so the page a stretch of extracted text came from
# is decidable by reading the text alone. THREE pages rather than two, and the
# MIDDLE one carries no text operators: a two-page fixture cannot tell a page
# INDEX from a page NUMBER, and without a blank page in the middle it cannot
# tell either of them from a count of the pages that produced text. The
# extractor drops empty pages, so the third page's text belongs to page 3 and
# to no other number.
PAGED_CONTENTS = (
b"BT /F1 12 Tf 20 160 Td (Side en om helning) Tj ET\n",
b"20 20 160 160 re S\n",
b"BT /F1 12 Tf 20 160 Td (Side tre om utkiling) Tj ET\n",
)
# Two fonts and three sizes on one page: a 20pt bold title, a 14pt bold
# subheading, and 10pt regular body. The PDF format carries no notion of a
# heading at all -- a heading in a PDF is a typographic fact, which is why the
# font-aware reader has to infer one -- so a fixture for that reader must state
# the typography and nothing else. The body is the majority of the characters,
# which is what gives the reader a body size to compare against.
FONT_HEADING_CONTENT = (
b"BT /F2 20 Tf 50 700 Td (Generelle tekniske krav) Tj ET\n"
b"BT /F1 10 Tf 50 670 Td (Utkilingen skal ha helning 1:15.) Tj ET\n"
b"BT /F2 14 Tf 50 640 Td (Merking) Tj ET\n"
b"BT /F1 10 Tf 50 610 Td (Kravet gjelder alle veiklasser.) Tj ET\n"
)
# The SAME typography as `FONT_HEADING_CONTENT`, over a document that numbers
# its own chapters. It exists for the heading RESERVE and for nothing else: the
# reserve reads typography only where Arm D's outline gate admits no run, so
# proving it stays silent needs a document where both signals are present and
# only one of them may be used. Three ascending integers at line start, which
# is a run at the build default's minimum of 3.
NUMBERED_FONT_CONTENT = (
b"BT /F2 20 Tf 50 700 Td (Forord) Tj ET\n"
b"BT /F1 10 Tf 50 670 Td (1 Generelle krav) Tj ET\n"
b"BT /F1 10 Tf 50 640 Td (Utkilingen skal ha helning 1:15.) Tj ET\n"
b"BT /F1 10 Tf 50 610 Td (2 Merking) Tj ET\n"
b"BT /F1 10 Tf 50 580 Td (Kravet gjelder alle veiklasser.) Tj ET\n"
b"BT /F1 10 Tf 50 550 Td (3 Vedlegg) Tj ET\n"
b"BT /F1 10 Tf 50 520 Td (Vedlegget er eget oppslag.) Tj ET\n"
)
# THREE pages and a THREE-LEVEL `/Outlines` tree, which is the shape the
# bookmark arm has to be proved against. A two-level tree cannot tell "the
# level the node declares" apart from "one below the root", and a
# one-page fixture cannot tell a page-local line offset from a document-wide
# one -- the arm's whole risk is the bridge from (page, y) to a line index.
#
# The third page carries FOUR lines and its second bookmark points at the
# THIRD of them, so a bridge that resolved to the page and stopped would put
# the mark two lines early and still look like it worked.
OUTLINED_CONTENTS = (
b"BT /F1 12 Tf 20 170 Td (1 Grunnlag) Tj ET\n"
b"BT /F1 12 Tf 20 150 Td (Innledende tekst om grunnlaget.) Tj ET\n",
b"BT /F1 12 Tf 20 170 Td (1.1 Omfang) Tj ET\n"
b"BT /F1 12 Tf 20 150 Td (Omfanget dekker hele arbeidet.) Tj ET\n",
b"BT /F1 12 Tf 20 170 Td (1.2 Krav) Tj ET\n"
b"BT /F1 12 Tf 20 150 Td (Kravet gjelder alle klasser.) Tj ET\n"
b"BT /F1 12 Tf 20 130 Td (1.2.1 Materialer) Tj ET\n"
b"BT /F1 12 Tf 20 110 Td (Materialene skal vaere godkjente.) Tj ET\n",
)
#: `(title, level, page index, /XYZ top)`. The `top` values are the PDF's own
#: bottom-up user space: 185 sits above the first line of a page and 150 above
#: its third, which is what makes the third-line mark a statement rather than
#: a coincidence.
OUTLINED_TREE = (
("1 Grunnlag", 1, 0, 185),
("1.1 Omfang", 2, 1, 185),
("1.2 Krav", 2, 2, 185),
("1.2.1 Materialer", 3, 2, 150),
)
#: TWO bookmarks whose destinations resolve to the SAME line, mirroring what
#: R761 carries: its tree's root node `R761 Prosesskoden` and the node
#: `SVV - Forside` both land on line 0. Measured on that document, 2 763 nodes
#: entered the bridge and 2 762 marks came out with `unresolved` at 0 -- the
#: difference was a dict keyed on the line index, dropping the second node with
#: nothing counting it. A one-bookmark-per-line fixture cannot see that.
COLLISION_CONTENTS = (
b"BT /F1 12 Tf 20 170 Td (R761 Prosesskoden) Tj ET\n"
b"BT /F1 12 Tf 20 150 Td (Innledende tekst om grunnlaget.) Tj ET\n",
)
#: Both point at `/XYZ 20 185`, which is above the page's first line.
COLLISION_TREE = (
("R761 Prosesskoden", 1, 0, 185),
("SVV - Forside", 2, 0, 185),
)
#: One resolvable bookmark and one whose `/Dest` names an object that is not a
#: page. A PDF in the wild carries these; R761 carries none of them, so
#: without this fixture the "drop it, count it, do not fabricate a boundary"
#: branch would ship having never run.
BROKEN_DEST_CONTENT = (
b"BT /F1 12 Tf 20 170 Td (1 Grunnlag) Tj ET\n"
b"BT /F1 12 Tf 20 150 Td (Innledende tekst om grunnlaget.) Tj ET\n",
)
def build_outlined_pdf(
contents: tuple[bytes, ...],
tree: tuple[tuple[str, int, int, int], ...],
*,
broken_dest: bool = False,
) -> bytes:
"""`build_paged_pdf` plus a hand-laid `/Outlines` tree in the catalog.
The tree is written from the flat `(title, level, page, top)` list rather
than from a nested literal, because `/First`, `/Last`, `/Next`, `/Prev` and
`/Parent` all have to agree with each other and with the level column --
a hand-written nest gets one of them wrong silently and the reader then
reports a level the document never declared.
`broken_dest` appends one extra item whose `/Dest` names the FONT object
instead of a page. It is a valid indirect reference to a real object that
is not a page, which is the failure a reader has to survive.
"""
count = len(contents)
page_numbers = [3 + 2 * index for index in range(count)]
font_number = 3 + 2 * count
outlines_number = font_number + 1
items = list(tree) + ([("Uoppl\u00f8selig", 1, -1, 185)] if broken_dest else [])
item_numbers = [outlines_number + 1 + index for index in range(len(items))]
kids = b" ".join(str(number).encode() + b" 0 R" for number in page_numbers)
objects = [
b"<< /Type /Catalog /Pages 2 0 R /Outlines " + str(outlines_number).encode() + b" 0 R >>",
b"<< /Type /Pages /Kids [" + kids + b"] /Count " + str(count).encode() + b" >>",
]
for index, content in enumerate(contents):
stream_number = page_numbers[index] + 1
objects.append(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Contents "
+ str(stream_number).encode()
+ b" 0 R /Resources << /Font << /F1 "
+ str(font_number).encode()
+ b" 0 R >> >> >>"
)
objects.append(
b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"endstream"
)
objects.append(
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
)
# The parent of an item is the last item seen at the level above it; its
# previous sibling is the last item seen at its OWN level under that same
# parent. Both are resolved in one forward pass so the links cannot drift.
parents: list[int | None] = []
last_at_level: dict[int, int] = {}
siblings: list[int | None] = []
children: dict[int, list[int]] = {}
roots: list[int] = []
for position, (_, level, _, _) in enumerate(items):
parent = last_at_level.get(level - 1) if level > 1 else None
parents.append(parent)
previous = None
group = children.setdefault(parent, []) if parent is not None else roots
if group:
previous = group[-1]
siblings.append(previous)
group.append(position)
last_at_level[level] = position
for deeper in [key for key in last_at_level if key > level]:
del last_at_level[deeper]
def _ref(position: int | None) -> bytes:
return b"" if position is None else str(item_numbers[position]).encode() + b" 0 R"
root_body = b"<< /Type /Outlines /Count " + str(len(items)).encode() + b" >>"
if roots:
root_body = (
b"<< /Type /Outlines /First "
+ _ref(roots[0])
+ b" /Last "
+ _ref(roots[-1])
+ b" /Count "
+ str(len(items)).encode()
+ b" >>"
)
objects.append(root_body)
for position, (title, _, page, top) in enumerate(items):
parent = parents[position]
parent_ref = _ref(parent) if parent is not None else str(outlines_number).encode() + b" 0 R"
group = children.get(parent, []) if parent is not None else roots
index_in_group = group.index(position)
body = b"<< /Title (" + _pdf_text(title) + b") /Parent " + parent_ref
if index_in_group > 0:
body += b" /Prev " + _ref(group[index_in_group - 1])
if index_in_group + 1 < len(group):
body += b" /Next " + _ref(group[index_in_group + 1])
own = children.get(position, [])
if own:
body += b" /First " + _ref(own[0]) + b" /Last " + _ref(own[-1])
body += b" /Count " + str(len(own)).encode()
target = str(font_number).encode() if page < 0 else str(page_numbers[page]).encode()
body += b" /Dest [" + target + b" 0 R /XYZ 20 " + str(top).encode() + b" 0] >>"
objects.append(body)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for number, item in enumerate(objects, start=1):
offsets.append(len(out))
out += str(number).encode() + b" 0 obj\n" + item + b"\nendobj\n"
xref_at = len(out)
size = str(len(objects) + 1).encode()
out += b"xref\n0 " + size + b"\n0000000000 65535 f \n"
for offset in offsets:
out += ("%010d 00000 n \n" % offset).encode()
out += b"trailer\n<< /Size " + size + b" /Root 1 0 R >>\n"
out += b"startxref\n" + str(xref_at).encode() + b"\n%%EOF\n"
return bytes(out)
def _pdf_text(value: str) -> bytes:
"""A PDF literal string in WinAnsi, with the three delimiters escaped."""
raw = value.encode("cp1252")
for character in (b"\\", b"(", b")"):
raw = raw.replace(character, b"\\" + character)
return raw
def build_two_font_pdf(content: bytes) -> bytes:
"""A one-page PDF whose resources declare BOTH a regular and a bold font.
Separate from `build_paged_pdf` rather than a parameter on it: that builder
emits exactly one font object and every existing fixture's bytes depend on
its object numbering. A second font changes the numbering, so sharing the
code would mean regenerating files whose whole value is that they have not
moved.
The page is Letter-sized rather than the 200x200 the other fixtures use,
because a 20pt line of this length does not fit inside 200 points and a
character laid outside the page box is not one a reader has to see.
"""
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Contents 4 0 R "
b"/Resources << /Font << /F1 5 0 R /F2 6 0 R >> >> >>",
b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"endstream",
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>",
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica-Bold /Encoding /WinAnsiEncoding >>",
]
out = bytearray(b"%PDF-1.4\n")
offsets = []
for number, body in enumerate(objects, start=1):
offsets.append(len(out))
out += str(number).encode() + b" 0 obj\n" + body + b"\nendobj\n"
xref_at = len(out)
size = str(len(objects) + 1).encode()
out += b"xref\n0 " + size + b"\n0000000000 65535 f \n"
for offset in offsets:
out += ("%010d 00000 n \n" % offset).encode()
out += b"trailer\n<< /Size " + size + b" /Root 1 0 R >>\n"
out += b"startxref\n" + str(xref_at).encode() + b"\n%%EOF\n"
return bytes(out)
def build_pdf(content: bytes) -> bytes:
"""Assemble a one-page PDF around `content` as the page content stream."""
return build_paged_pdf((content,))
def build_paged_pdf(contents: tuple[bytes, ...]) -> bytes:
"""Assemble a PDF with one page per entry of `contents`.
The object numbering is laid out first and the xref offsets computed from
the emitted bytes, exactly as the single-page form did -- the fixture stays
a hand-written statement about the format rather than a library's output.
"""
count = len(contents)
# 1 catalog, 2 pages, then one page object and one content stream per page,
# and the shared font last.
page_numbers = [3 + 2 * index for index in range(count)]
font_number = 3 + 2 * count
kids = b" ".join(str(number).encode() + b" 0 R" for number in page_numbers)
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [" + kids + b"] /Count " + str(count).encode() + b" >>",
]
for index, content in enumerate(contents):
stream_number = page_numbers[index] + 1
objects.append(
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Contents "
+ str(stream_number).encode()
+ b" 0 R /Resources << /Font << /F1 "
+ str(font_number).encode()
+ b" 0 R >> >> >>"
)
objects.append(
b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"endstream"
)
objects.append(
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
)
out = bytearray(b"%PDF-1.4\n")
offsets = []
for number, body in enumerate(objects, start=1):
offsets.append(len(out))
out += str(number).encode() + b" 0 obj\n" + body + b"\nendobj\n"
xref_at = len(out)
size = str(len(objects) + 1).encode()
out += b"xref\n0 " + size + b"\n0000000000 65535 f \n"
for offset in offsets:
out += ("%010d 00000 n \n" % offset).encode()
out += b"trailer\n<< /Size " + size + b" /Root 1 0 R >>\n"
out += b"startxref\n" + str(xref_at).encode() + b"\n%%EOF\n"
return bytes(out)
# --- office containers -------------------------------------------------------
#
# A fixed timestamp on every member, because a zip records mtime and the whole
# point is a byte-reproducible file: without it the fixture would differ on
# every regeneration and `git diff --quiet` could never be the check.
_ZIP_DATE = (2020, 1, 1, 0, 0, 0)
_XML = ''
# `word/styles.xml` IS REQUIRED, not decoration. Measured during planning: the
# same document WITHOUT a styles part extracts as plain text with no heading
# marker at all, so a fixture lacking it would pin the body and silently pin
# nothing about structure -- which is the half the segment proposer reads.
_DOCX_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _XML
+ ''
+ ''
+ "",
"word/_rels/document.xml.rels": _XML
+ ''
+ ''
+ "",
"word/styles.xml": _XML
+ ''
+ ''
+ "",
"word/document.xml": _XML
+ ''
+ 'Krav til helning'
+ "60 og 70 1:15"
+ "",
}
# The same document MINUS the styles part. A negative control, committed rather
# than described: it is what proves the styles part is load-bearing, and a
# claim of that kind that nothing runs is a claim that decays.
_DOCX_NO_STYLES_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _DOCX_PARTS["_rels/.rels"],
"word/document.xml": _DOCX_PARTS["word/document.xml"],
}
# A minimal SpreadsheetML workbook: one sheet, a heading row and a label/value
# row, mirroring what the PDF fixture does for its format.
#
# THE SHARED STRING TABLE IS NOT A STYLE CHOICE. The first attempt used inline
# strings (`t="inlineStr"`), which is valid SpreadsheetML and which the
# converter reads as EMPTY CELLS -- the sheet name survived and every value
# vanished, with exit code 0 and no warning. A `dimension` element and a shared
# string table are what make the values arrive. This is the same class of
# defect as the missing `styles.xml`: structurally valid input, silently
# reduced output, nothing anywhere saying so.
_XLSX_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _XML
+ ''
+ ''
+ "",
"xl/_rels/workbook.xml.rels": _XML
+ ''
+ ''
+ ''
+ "",
"xl/workbook.xml": _XML
+ ''
+ '',
"xl/sharedStrings.xml": _XML
+ ''
+ "Krav til helning60 og 701:15",
"xl/worksheets/sheet1.xml": _XML
+ ''
+ ''
+ '0
'
+ '12
'
+ "",
}
# A price sheet, and the negative control for it, in ONE workbook.
#
# Sheet 1 mirrors the shape measured on the K2 price sheet: row 1 carries a
# single title cell, so the table's HEADER ROW names one column while the rows
# below it carry three. That is what makes a reader see one column and a
# whitespace carpet where the source has a label and an amount. Column B also
# holds one long prose cell, which is what makes the simple-table writer pad
# every other row in that column out to its width.
#
# Sheet 2 is the negative control in the same file: one column in the SOURCE,
# so there are no columns to recover and nothing for a fix to invent.
#
# THREE NUMERIC CELLS AND ONE THAT ONLY LOOKS NUMERIC. `5647500` and `250000`
# are stored as numbers and are integral; `12.5` is stored as a number and is
# not; `92.0` is a SHARED STRING. The converter renders the first two as
# `5647500.0` and `250000.0` and the last one as `92.0` -- identical output for
# a number and for text, which is the whole reason the shared string table is
# consulted before any of them is rewritten.
_PRISARK_STRINGS = (
"Prisskjema",
"Post",
"Beskrivelse",
"Sum",
"01",
"Rigging og drift av byggeplass, medregnet alt som ikke er priset "
"spesifikt nedenfor og alt som er innkalkulert i de angitte prisene",
"02",
"Andel",
"03",
"92.0",
"Notat",
"Ingen kolonner her",
"Sum ikke oppgitt",
# A cell whose own text contains a pipe and a number. The converter escapes
# the pipe inside a pipe table, and the escape is what the rewrite's
# delimiter test has to survive: a `5.0` INSIDE a cell is not a cell.
"Kode 4 | 5.0",
"04",
)
_PRISARK_SHEET1 = (
'0
'
'12'
'3
'
'45'
'5647500
'
'67'
'12.5
'
'89'
'250000
'
'1413
'
)
_PRISARK_SHEET2 = (
'10
'
'11
'
'12
'
)
def _sheet(dimension: str, rows: str) -> str:
return (
_XML
+ ''
+ f''
+ rows
+ ""
)
_PRISARK_PARTS = {
"[Content_Types].xml": _XML
+ ''
+ ''
+ ''
+ ''
+ ''
+ ''
+ ''
+ "",
"_rels/.rels": _XLSX_PARTS["_rels/.rels"],
"xl/_rels/workbook.xml.rels": _XML
+ ''
+ ''
+ ''
+ ''
+ "",
"xl/workbook.xml": _XML
+ ''
+ ''
+ '',
"xl/sharedStrings.xml": _XML
+ ''
+ "".join(f"{value}" for value in _PRISARK_STRINGS)
+ "",
"xl/worksheets/sheet1.xml": _sheet("A1:C6", _PRISARK_SHEET1),
"xl/worksheets/sheet2.xml": _sheet("A1:A3", _PRISARK_SHEET2),
}
# A sheet with an EMPTY ROW IN THE MIDDLE, which the converter renders as a
# pipe row of nothing but spaces. That row looks exactly like a table's own
# separator line to any rule that reads the line rather than its position --
# and swallowing it renumbers every row after it, silently, for the whole
# sheet. Measured on the K2 price sheet before this fixture existed: 8 empty
# rows, and the last row reported as 92 when the workbook says 100.
_TOMRAD_STRINGS = ("Rad en", "Rad to", "Rad fire")
_TOMRAD_SHEET = (
'0
'
'1
'
'
'
'2
'
)
_TOMRAD_PARTS = {
"[Content_Types].xml": _XLSX_PARTS["[Content_Types].xml"],
"_rels/.rels": _XLSX_PARTS["_rels/.rels"],
"xl/_rels/workbook.xml.rels": _XLSX_PARTS["xl/_rels/workbook.xml.rels"],
"xl/workbook.xml": _XML
+ ''
+ '',
"xl/sharedStrings.xml": _XML
+ ''
+ "".join(f"{value}" for value in _TOMRAD_STRINGS)
+ "",
"xl/worksheets/sheet1.xml": _sheet("A1:A4", _TOMRAD_SHEET),
}
def build_ooxml(parts: dict[str, str]) -> bytes:
"""Zip the parts with a fixed timestamp and no compression variance.
A constant `date_time` on every member is what makes the output
reproducible: a zip records mtime, so the default `ZipFile.writestr` would
stamp the current time and the fixture would differ on every run --
which would make `git diff --quiet` useless as the regeneration check.
"""
out = io.BytesIO()
with zipfile.ZipFile(out, "w", compression=zipfile.ZIP_DEFLATED) as archive:
for name, payload in parts.items():
info = zipfile.ZipInfo(name, date_time=_ZIP_DATE)
info.compress_type = zipfile.ZIP_DEFLATED
archive.writestr(info, payload)
return out.getvalue()
if __name__ == "__main__":
for name, content in (
("two-line-krav.pdf", KRAV_CONTENT),
("no-text-layer.pdf", NO_TEXT_CONTENT),
):
(HERE / name).write_bytes(build_pdf(content))
print(f"wrote {name}")
(HERE / "three-page-krav.pdf").write_bytes(build_paged_pdf(PAGED_CONTENTS))
print("wrote three-page-krav.pdf")
(HERE / "font-heading-krav.pdf").write_bytes(build_two_font_pdf(FONT_HEADING_CONTENT))
print("wrote font-heading-krav.pdf")
(HERE / "numbered-font-krav.pdf").write_bytes(build_two_font_pdf(NUMBERED_FONT_CONTENT))
print("wrote numbered-font-krav.pdf")
(HERE / "outlined-krav.pdf").write_bytes(build_outlined_pdf(OUTLINED_CONTENTS, OUTLINED_TREE))
print("wrote outlined-krav.pdf")
(HERE / "outline-collision.pdf").write_bytes(
build_outlined_pdf(COLLISION_CONTENTS, COLLISION_TREE)
)
print("wrote outline-collision.pdf")
(HERE / "outline-broken-dest.pdf").write_bytes(
build_outlined_pdf(BROKEN_DEST_CONTENT, (("1 Grunnlag", 1, 0, 185),), broken_dest=True)
)
print("wrote outline-broken-dest.pdf")
for name, parts in (
("two-line-krav.docx", _DOCX_PARTS),
("no-styles-krav.docx", _DOCX_NO_STYLES_PARTS),
("two-line-krav.xlsx", _XLSX_PARTS),
("prisark.xlsx", _PRISARK_PARTS),
("tomrad.xlsx", _TOMRAD_PARTS),
):
(HERE / name).write_bytes(build_ooxml(parts))
print(f"wrote {name}")