llm-ingestion-okf/tools/okf_witness.py
Kjell Tore Guttormsen d1de1a6a9d
fix(docs): N101 ships 568 tbx: tags, not 574
Counted three ways over `N101-2025-860031.json`, none of them importing this
package: raw `tbx:` substring 568, regex over `"tag": "tbx:` 568, node
traversal 568 across 10 distinct tags (94 x 6 + 1 x 4), out of 7 374 tags in
the delivery. The number is corrected in all four places that published it --
CLAUDE.md, `tools/okf_witness.py`'s `_sts_role_json` docstring,
`tests/test_accounting_gate.py` and the round's report -- and the test from
the previous commit now counts it rather than repeating it.

The one remaining `574` in the tree is `docs/2026-09-11-k3-runde8...md:89`
("11 016 and 6 574"), a character count of two documents and a different
number.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-19 08:35:50 +02:00

1353 lines
49 KiB
Python

"""The independent witness: what a SOURCE file holds, counted by the format's rules.
This module is the fasit side of the content-accounting gate
(`tools/okf_accounting_gate.py`). It answers one question per file -- "how many
of each element does this source carry?" -- and it answers it WITHOUT this
package: no `llm_ingestion_okf` module is imported, directly or through a
helper, and `tests/test_accounting_gate.py` proves that on the live import
graph rather than by searching the text. A fasit computed by the reader it is
meant to judge is the reader agreeing with itself.
Each counter reads the container the way the format defines it:
- XML (NISO-STS): `xml.etree.ElementTree` straight on the bytes.
- STS JSON twin: the publisher's own node tree (`standardContent`, nodes with
`e`/`t`/`x`), walked with the same element roles as the XML.
- docx / pptx / xlsx / odt: the zip members' own XML.
- PDF: pdfplumber OBJECTS (pages, image placements) and, as a second witness,
poppler (`pdfinfo`, `pdfimages -list`). pdfplumber is also what the reader
extracts text with, which is why the gate never trusts a PDF count that the
poppler side does not repeat.
- HTML: `html.parser` from the stdlib.
- md / txt / csv / json / rtf: stdlib line, csv and json readers, and a
control-word scan for rtf.
The ELEMENT VOCABULARY is part of the gate's contract: a build that declares
an inventory must use these names, per file type, or it is not comparable.
Each name is defined where it is counted, and a type counts only what that
format actually carries.
Image REFERENCES are resolved here too, because the accounting has to know
which inbox files a document points at: a relative reference is taken against
the document's own directory, never above it, and an STS reference that is
not found there is looked for as `graphics/<basename>` -- the layout the
publisher's STS delivery ships in. That is a fact about the delivery format,
written down here, not borrowed from the reader.
"""
from __future__ import annotations
import csv
import io
import json
import re
import shutil
import subprocess
import zipfile
from collections.abc import Iterable, Iterator, Mapping
from dataclasses import dataclass, field
from html.parser import HTMLParser
from pathlib import Path, PurePosixPath
from typing import Any
from xml.etree import ElementTree as ET
WITNESS_VERSION = 1
#: Pointer kinds a document can hold for an image.
LOCAL = "local"
REMOTE = "remote"
EMBEDDED = "embedded"
class WitnessRefused(Exception):
"""The witness will not read this file (for example a DOCTYPE)."""
@dataclass(frozen=True)
class ImageRef:
"""One image a document declares.
`target` is the inbox-relative POSIX path of the file a LOCAL reference
resolves to, or None when it resolves to nothing inside the document's
directory.
"""
kind: str
ref: str
target: str | None = None
class Count:
"""Elements of one file: how many of each, and THE TEXT OF EACH.
The text is what makes the gate a judge. A count alone can only be
compared with another count, so a report claiming an element was carried
could never be checked against the bundle; with the element's own text
the gate looks for it and says whether it is there.
An element that carries no text of its own (a picture, a spreadsheet's
sheet) gets the empty string, and the gate reports it as one it cannot
check rather than as one that passed.
"""
def __init__(self, vocabulary: Iterable[str]) -> None:
self.vocabulary = tuple(vocabulary)
self.counts: dict[str, int] = dict.fromkeys(self.vocabulary, 0)
self.texts: dict[str, list[list[str]]] = {name: [] for name in self.vocabulary}
def add(self, role: str, *pieces: str) -> list[str]:
"""Count one element and keep the PIECES of text it is made of.
Pieces, not one joined string: a reader writes a heading's marker and
a picture's pointer block between the parts of a container, so a
section's text is not a contiguous run in the bundle even when every
word of it is there. Each piece is looked for on its own.
"""
if role not in self.counts:
raise KeyError(f"{role!r} is not in this format's vocabulary")
kept = [piece for piece in pieces if piece and piece.strip()]
self.counts[role] += 1
self.texts[role].append(kept)
return kept
@dataclass
class Inventory:
"""What one source file holds, element type by element type."""
source_file: str
suffix: str
witness: str
elements: dict[str, int] = field(default_factory=dict)
images: list[ImageRef] = field(default_factory=list)
texts: dict[str, list[list[str]]] = field(default_factory=dict)
@property
def total(self) -> int:
return sum(self.elements.values())
def to_json(self) -> dict[str, object]:
return {
"suffix": self.suffix,
"witness": self.witness,
"elements": dict(sorted(self.elements.items())),
"texts": {
name: [list(pieces) for pieces in values]
for name, values in sorted(self.texts.items())
},
"images": [
{"kind": ref.kind, "ref": ref.ref, "target": ref.target} for ref in self.images
],
}
def _normal(text: str) -> str:
"""One space between words: a reader re-wraps, and a piece has to survive
that to be looked for at all."""
return " ".join(text.split())
def _local(tag: str) -> str:
return tag.rsplit("}", 1)[-1] if "}" in tag else tag.split(":")[-1]
def _is_remote(ref: str) -> bool:
return bool(re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*:", ref)) or ref.startswith("//")
def resolve_local(inbox: Path, document: Path, ref: str, *, sts: bool = False) -> str | None:
"""The inbox-relative path a LOCAL reference names, or None.
Contained in the document's own directory: an absolute path or one that
climbs above that directory resolves to nothing.
"""
base = document.parent
candidates = [ref]
if sts:
candidates.append(f"graphics/{PurePosixPath(ref).name}")
for candidate in candidates:
pure = PurePosixPath(candidate)
if pure.is_absolute() or ".." in pure.parts:
continue
target = base.joinpath(*pure.parts)
if target.is_file():
return target.relative_to(inbox).as_posix()
return None
# --- markdown / text ---------------------------------------------------------
_FENCE_OPEN = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$")
_ATX_LINE = re.compile(r"^ {0,3}#{1,6}(\s|$)")
_DELIMITER_ROW = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
_MD_IMAGE = re.compile(r"!\[[^\]]*\]\(\s*<?([^)\s>]+)>?[^)]*\)")
def _unfenced(lines: list[str]) -> tuple[list[str | None], list[list[str]]]:
"""Lines with fenced ones replaced by None, and one block per OPENING fence.
CommonMark SS 4.5 in the parts that decide which lines are fenced: up to
three leading spaces, a backtick info string may not hold a backtick, the
closing fence is the same character and at least as long, and an unclosed
fence runs to the end of the text. The blocks are kept apart per opener,
because two fences may stand on consecutive lines and a run of fenced
lines would then read as one.
"""
out: list[str | None] = []
blocks: list[list[str]] = []
opener: str | None = None
for line in lines:
if opener is None:
match = _FENCE_OPEN.match(line)
if match and not (match.group(1)[0] == "`" and "`" in match.group(2)):
opener = match.group(1)
blocks.append([line])
out.append(None)
continue
out.append(line)
continue
out.append(None)
blocks[-1].append(line)
stripped = line.strip()
if (
stripped
and set(stripped) == {opener[0]}
and len(stripped) >= len(opener)
and len(line) - len(line.lstrip(" ")) <= 3
):
opener = None
return out, blocks
MARKDOWN = ("code_block", "heading", "image", "paragraph", "table", "table_row")
def count_markdown(text: str) -> tuple[Count, list[str]]:
"""heading: ATX lines outside a fence. table: a pipe row followed by a
delimiter row. table_row: the body rows under it. image: `![..](..)`
outside a fence. code_block: a fence. paragraph: a run of non-blank lines
outside a fence that holds none of the above.
Each element's text is the line, or the lines, it is made of."""
lines, blocks = _unfenced(text.split("\n"))
count = Count(MARKDOWN)
refs: list[str] = []
in_table = False
paragraph: list[str] = []
table_header: str | None = None
table_rows: list[str] = []
delimiter_rows: set[int] = set()
def _close_table() -> None:
nonlocal table_header
if table_header is not None:
count.add("table", table_header, *table_rows)
table_header = None
table_rows.clear()
def close_paragraph() -> None:
if paragraph:
count.add("paragraph", *paragraph)
paragraph.clear()
for index, line in enumerate(lines):
if index in delimiter_rows:
continue
if line is None or not line.strip():
if in_table:
in_table = False
_close_table()
close_paragraph()
continue
if in_table:
if "|" in line:
count.add("table_row", line)
table_rows.append(line)
continue
in_table = False
_close_table()
following = lines[index + 1] if index + 1 < len(lines) else None
if "|" in line and following is not None and _DELIMITER_ROW.match(following):
_close_table()
table_header = line
in_table = True
delimiter_rows.add(index + 1) # the delimiter row is not a body row
close_paragraph()
continue
if _ATX_LINE.match(line):
count.add("heading", line)
close_paragraph()
continue
found = _MD_IMAGE.findall(line)
rest = _MD_IMAGE.sub("", line) if found else line
for ref in found:
count.add("image")
refs.append(ref)
if found and not rest.strip():
close_paragraph()
continue
paragraph.append(rest)
close_paragraph()
_close_table()
for block in blocks:
count.add("code_block", "\n".join(block))
return count, refs
TEXT = ("line", "paragraph")
def count_text(text: str) -> Count:
"""paragraph: a run of non-blank lines. line: a non-blank line."""
count = Count(TEXT)
block: list[str] = []
for line in [*text.split("\n"), ""]:
if line.strip():
count.add("line", line)
block.append(line)
elif block:
count.add("paragraph", *block)
block = []
return count
CSV = ("cell", "header_cell", "row")
def count_csv(text: str) -> Count:
"""header_cell: cells of the first row. row / cell: every row after it."""
count = Count(CSV)
rows = [row for row in csv.reader(io.StringIO(text)) if row]
for position, row in enumerate(rows):
if position == 0:
for value in row:
count.add("header_cell", value)
continue
for value in row:
count.add("cell", value)
count.add("row", *row)
return count
JSON = ("key", "value")
def count_json(text: str) -> Count:
"""key: an object member. value: a leaf (string, number, boolean, null).
A key's text is the key; a leaf's text is the leaf as JSON writes it,
which is the form a bundle carrying the document verbatim holds."""
count = Count(JSON)
def walk(node: object) -> None:
if isinstance(node, dict):
for key, child in node.items():
count.add("key", key)
walk(child)
elif isinstance(node, list):
for child in node:
walk(child)
else:
count.add("value", json.dumps(node, ensure_ascii=False))
walk(json.loads(text))
return count
# --- html --------------------------------------------------------------------
HTML = ("cell", "heading", "image", "list_item", "paragraph", "table")
#: Tags that close themselves: an unclosed `<img>` must not swallow the rest
#: of the document as its own text.
_HTML_VOID = frozenset(
{"img", "br", "hr", "meta", "link", "input", "col", "area", "base", "wbr", "source"}
)
class _HtmlCounter(HTMLParser):
"""heading: h1-h6. paragraph: p. list_item: li. table: table. cell: td,
th. image: img. An element's text is the text between its own tags."""
_ROLES = {
**{f"h{level}": "heading" for level in range(1, 7)},
"p": "paragraph",
"li": "list_item",
"table": "table",
"td": "cell",
"th": "cell",
"img": "image",
}
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.count = Count(HTML)
self.refs: list[str] = []
self._open: list[tuple[str, str, list[str]]] = []
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
role = self._ROLES.get(tag)
if tag == "img":
if role is not None:
self.count.add(role)
self.refs.append(dict(attrs).get("src") or "")
return
if role is None:
return
self._open.append((tag, role, []))
def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
self.handle_starttag(tag, attrs)
def handle_endtag(self, tag: str) -> None:
if tag in _HTML_VOID or self._ROLES.get(tag) is None:
return
for position in range(len(self._open) - 1, -1, -1):
if self._open[position][0] == tag:
_, role, pieces = self._open.pop(position)
kept = self.count.add(role, *pieces)
for outer in self._open:
outer[2].extend(kept)
return
def handle_data(self, data: str) -> None:
if self._open:
self._open[-1][2].append(_normal(data))
def finish(self) -> None:
"""Whatever the document left open still counts, innermost first."""
while self._open:
_, role, pieces = self._open.pop()
kept = self.count.add(role, *pieces)
for outer in self._open:
outer[2].extend(kept)
# --- xml / sts ---------------------------------------------------------------
STS_ROLES = (
"cell",
"citation",
"figure",
"figure_caption",
"footnote",
"image",
"list_item",
"math",
"paragraph",
"section",
"section_label",
"table",
"table_label",
"title",
)
def _sts_role_xml(tag: str, parent: str | None, grandparent: str | None) -> str | None:
"""An STS XML element's accounting role.
Written for the XML delivery ALONE. Until 2026-09-18 one function served
both deliveries, so row 5 -- "two witnesses agree" -- could not see a hole
in it: a role missing here was missing there, and the two agreed on a
number neither of them should have produced (independent review, M-2).
In this delivery a section's label is `sec/label` and a table's label is
`table-wrap/label`. Measured on R761 2026-09-18: `sec/label` 7 714,
`sec/title/label` **0** -- the nested placement is a fact about the JSON
delivery and does not belong in this map.
"""
if tag == "sec":
return "section"
if tag == "title" and parent == "sec":
return "title"
if tag == "label" and parent == "sec":
return "section_label"
if tag == "label" and parent == "table-wrap":
return "table_label"
if tag == "caption" and parent == "table" and grandparent == "table-wrap":
return "table_label"
if tag == "caption" and parent == "fig":
return "figure_caption"
if tag == "fig":
return "figure"
if tag == "mixed-citation":
return "citation"
if tag == "math":
return "math"
if tag == "p":
return "paragraph"
if tag == "table-wrap":
return "table"
if tag in ("td", "th"):
return "cell"
if tag == "list-item":
return "list_item"
if tag in ("graphic", "inline-graphic"):
return "image"
if tag == "fn":
return "footnote"
return None
def _sts_role_json(tag: str, parent: str | None, grandparent: str | None) -> str | None:
"""The same roles, read from the publisher's JSON node tree.
Written apart from the XML map, because the publisher's two deliveries of
ONE document place the same text differently (measured on R761
Prosesskoden:2025, 2026-09-17):
- a section's label: XML `sec/label` on 7 714 sections; JSON `sec/label`
on 4 954 and `sec/title/label` on the 2 760 that carry a title.
- a table's label: XML `table-wrap/label` (10); JSON
`table-wrap/table/caption` (10).
Counted by tag alone, the two witnesses disagree by 2 760 and by 10 on
text both of them carry. The role, not the tag, is the unit.
The tag arrives through :func:`_local`, as the XML map's does: this
delivery writes a namespace as a PREFIX on the tag string, so `mml:math`
reached `tag == "math"` on nothing at all (H3, 74 of them in N200). The
known-negative for stripping it is N101's 568 `tbx:` tags, none of whose
local names is a role here.
`figcaption` is this delivery's own, measured and not assumed: NISO-STS
writes a figure's caption as `fig/caption`, and the publisher's JSON puts
it under the `graphic` instead -- 430 of them over the eight deliveries on
this machine, 0 reached by the `caption` clause above. It is deliberately
NOT in the XML map, which has never been measured carrying one.
"""
if tag == "sec":
return "section"
if tag == "title" and parent == "sec":
return "title"
if tag == "label" and parent == "sec":
return "section_label"
if tag == "label" and parent == "title" and grandparent == "sec":
return "section_label"
if tag == "label" and parent == "table-wrap":
return "table_label"
if tag == "caption" and parent == "table" and grandparent == "table-wrap":
return "table_label"
if tag == "caption" and parent == "fig":
return "figure_caption"
if tag == "figcaption":
return "figure_caption"
if tag == "fig":
return "figure"
if tag == "mixed-citation":
return "citation"
if tag == "math":
return "math"
if tag == "p":
return "paragraph"
if tag == "table-wrap":
return "table"
if tag in ("td", "th"):
return "cell"
if tag == "list-item":
return "list_item"
if tag in ("graphic", "inline-graphic"):
return "image"
if tag == "fn":
return "footnote"
return None
def count_sts_xml(data: bytes) -> tuple[Count, list[str], bool]:
"""Element roles of an STS document; `element` alone for other XML.
An element's text is its own subtree, whitespace-folded -- the form a
reader writing markdown produces, and the only form in which a container
section can be looked for at all."""
if b"<!DOCTYPE" in data:
raise WitnessRefused("a DOCTYPE is not parsed")
root = ET.fromstring(data)
sts = _local(root.tag) == "standard" or any(_local(el.tag) == "sec" for el in root.iter())
if not sts:
count = Count(("element",))
for node in root.iter():
pieces = [_normal(t) for t in node.itertext() if t.strip()]
count.add("element", *pieces)
return count, [], False
count = Count(STS_ROLES)
refs: list[str] = []
def walk(node: ET.Element, parent: str | None, grandparent: str | None) -> list[str]:
tag = _local(node.tag)
role = _sts_role_xml(tag, parent, grandparent)
pieces: list[str] = []
if node.text and node.text.strip():
pieces.append(_normal(node.text))
if role == "image":
href = next((value for key, value in node.attrib.items() if _local(key) == "href"), "")
refs.append(href)
for child in node:
pieces.extend(walk(child, tag, parent))
if child.tail and child.tail.strip():
pieces.append(_normal(child.tail))
if role is not None:
count.add(role, *pieces)
return pieces
walk(root, None, None)
return count, refs, True
def is_sts_json(data: bytes) -> bool:
"""Is this the publisher's JSON delivery of an STS document?
A standard shipped as JSON holds sections, titles, citations and tables.
Counted as generic JSON it holds keys and leaves: the container, not the
content -- and then no row can see that a citation left the bundle."""
if b'"standardContent"' not in data:
return False
try:
document = json.loads(data)
except ValueError:
return False
content = document.get("standardContent") if isinstance(document, dict) else None
return isinstance(content, dict) and isinstance(content.get("c"), list)
def count_sts_json(data: bytes) -> Count:
"""The same roles, read from the publisher's JSON node tree."""
document = json.loads(data)
count = Count(STS_ROLES)
def walk(node: Mapping[str, Any], parent: str | None, grandparent: str | None) -> list[str]:
pieces: list[str] = []
if node.get("t") and str(node["t"]).strip():
pieces.append(_normal(str(node["t"])))
body = node.get("x")
if not isinstance(body, dict):
return pieces
tag = _local(str(body.get("tag")))
role = _sts_role_json(tag, parent, grandparent)
for child in body.get("c") or []:
pieces.extend(walk(child, tag, parent))
if role is not None:
count.add(role, *pieces)
return pieces
for child in document["standardContent"]["c"]:
walk(child, None, None)
return count
# --- office zips -------------------------------------------------------------
_W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
_A = "{http://schemas.openxmlformats.org/drawingml/2006/main}"
_P = "{http://schemas.openxmlformats.org/presentationml/2006/main}"
_S = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
_XDR = "{http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing}"
_TEXT = "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}"
_TABLE = "{urn:oasis:names:tc:opendocument:xmlns:table:1.0}"
_DRAW = "{urn:oasis:names:tc:opendocument:xmlns:drawing:1.0}"
_HEADING_STYLE = re.compile(r"^(heading|overskrift|title|tittel)\s*\d*$", re.IGNORECASE)
def _text_of(node: ET.Element, tag: str) -> str:
return "".join(t.text or "" for t in node.iter(tag))
DOCX = (
"cell",
"comment",
"endnote",
"footnote",
"header_footer",
"heading",
"image",
"paragraph",
"table",
"text_box",
)
_DOCX_HEADER_FOOTER = re.compile(r"word/(header|footer)\d*\.xml")
def _docx_lines(para: ET.Element) -> list[str]:
"""A paragraph's text, cut where the document itself breaks a line.
`w:br` and `w:cr` are line boundaries in the format, so the two halves of
a broken paragraph can land in different places -- inside a grid table
they land on different rows, with other cells' text between them."""
lines = [""]
def walk(node: ET.Element) -> None:
for child in node:
# A text box holds its own paragraphs. Read as part of the
# paragraph that carries the box, its text is counted twice.
if child.tag == f"{_W}txbxContent":
continue
if child.tag == f"{_W}t":
lines[-1] += child.text or ""
elif child.tag in (f"{_W}br", f"{_W}cr"):
lines.append("")
walk(child)
walk(para)
return [_normal(line) for line in lines if line.strip()]
def count_docx(data: bytes) -> Count:
"""heading: a w:p whose style is a heading or title style. paragraph: any
other w:p with text. table: w:tbl. cell: w:tc. image: a:blip.
footnote: a w:footnote with a positive id."""
count = Count(DOCX)
parts: dict[str, ET.Element] = {}
with zipfile.ZipFile(io.BytesIO(data)) as archive:
root = ET.fromstring(archive.read("word/document.xml"))
for name in sorted(archive.namelist()):
if name in ("word/footnotes.xml", "word/endnotes.xml", "word/comments.xml") or (
_DOCX_HEADER_FOOTER.fullmatch(name)
):
parts[name] = ET.fromstring(archive.read(name))
# A text box's paragraphs are `w:p` in the body too: counted as prose they
# would be booked twice, so the box owns them and they are its pieces.
boxed: set[int] = set()
for box in root.iter(f"{_W}txbxContent"):
boxed.update(id(p) for p in box.iter(f"{_W}p"))
count.add("text_box", *[line for p in box.iter(f"{_W}p") for line in _docx_lines(p)])
for para in root.iter(f"{_W}p"):
if id(para) in boxed:
continue
style = para.find(f"{_W}pPr/{_W}pStyle")
lines = _docx_lines(para)
if style is not None and _HEADING_STYLE.match(style.get(f"{_W}val", "")):
count.add("heading", *lines)
elif lines:
count.add("paragraph", *lines)
for table in root.iter(f"{_W}tbl"):
count.add("table", *[line for p in table.iter(f"{_W}p") for line in _docx_lines(p)])
for cell in root.iter(f"{_W}tc"):
count.add("cell", *[line for p in cell.iter(f"{_W}p") for line in _docx_lines(p)])
for _ in root.iter(f"{_A}blip"):
count.add("image")
for name, part in parts.items():
if _DOCX_HEADER_FOOTER.fullmatch(name):
for para in part.iter(f"{_W}p"):
lines = _docx_lines(para)
if lines:
count.add("header_footer", *lines)
continue
role, tag = (
("footnote", f"{_W}footnote")
if name.endswith("footnotes.xml")
else ("endnote", f"{_W}endnote")
if name.endswith("endnotes.xml")
else ("comment", f"{_W}comment")
)
for note in part.iter(tag):
# A separator note carries id 0 and no document text.
if role != "comment" and int(note.get(f"{_W}id", "0")) <= 0:
continue
count.add(role, *[line for p in note.iter(f"{_W}p") for line in _docx_lines(p)])
return count
PPTX = ("cell", "hidden_slide", "image", "note", "paragraph", "slide", "table", "title")
def count_pptx(data: bytes) -> Count:
"""slide: ppt/slides/slideN.xml. title: a shape whose placeholder is a
title. paragraph: an a:p with text outside a table and outside a title.
table: a:tbl. cell: a:tc. image: p:pic."""
count = Count(PPTX)
with zipfile.ZipFile(io.BytesIO(data)) as archive:
slides = [n for n in archive.namelist() if re.fullmatch(r"ppt/slides/slide\d+\.xml", n)]
notes = [
n for n in archive.namelist() if re.fullmatch(r"ppt/notesSlides/notesSlide\d+\.xml", n)
]
for name in sorted(slides, key=_slide_order):
root = ET.fromstring(archive.read(name))
# `show="0"` is the deck saying this slide is not shown. Counted as
# an ordinary slide it is indistinguishable from one that is.
hidden = root.get("show") == "0"
count.add("hidden_slide" if hidden else "slide", *_pptx_lines(root))
for table in root.iter(f"{_A}tbl"):
count.add("table", *_pptx_lines(table))
for cell in root.iter(f"{_A}tc"):
count.add("cell", *_pptx_lines(cell))
for _ in root.iter(f"{_P}pic"):
count.add("image")
for shape in root.iter(f"{_P}sp"):
placeholder = shape.find(f"{_P}nvSpPr/{_P}nvPr/{_P}ph")
is_title = placeholder is not None and placeholder.get("type") in (
"title",
"ctrTitle",
)
texts = [
_normal(_text_of(p, f"{_A}t"))
for p in shape.iter(f"{_A}p")
if _text_of(p, f"{_A}t").strip()
]
if is_title and texts:
count.add("title", *texts)
else:
for text in texts:
count.add("paragraph", text)
for name in sorted(notes, key=_slide_order):
root = ET.fromstring(archive.read(name))
for line in _pptx_lines(root):
count.add("note", line)
return count
def _pptx_lines(node: ET.Element) -> list[str]:
"""One piece per a:p that holds text."""
return [
_normal(_text_of(p, f"{_A}t")) for p in node.iter(f"{_A}p") if _text_of(p, f"{_A}t").strip()
]
def _slide_order(name: str) -> tuple[int, str]:
match = re.search(r"(\d+)", name)
return (int(match.group(1)) if match else 0, name)
XLSX = ("cell", "formula", "hidden_sheet", "image", "row", "sheet")
def _shared_strings(archive: zipfile.ZipFile) -> list[str]:
if "xl/sharedStrings.xml" not in archive.namelist():
return []
root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
return [_normal(_text_of(item, f"{_S}t")) for item in root.iter(f"{_S}si")]
def _cell_text(cell: ET.Element, shared: list[str]) -> str:
inline = cell.find(f"{_S}is")
if inline is not None:
return _normal(_text_of(inline, f"{_S}t"))
value = cell.find(f"{_S}v")
raw = (value.text or "") if value is not None else ""
if cell.get("t") == "s":
try:
return shared[int(raw)]
except (ValueError, IndexError):
return ""
return _normal(raw)
def _hidden_sheets(archive: zipfile.ZipFile) -> set[str]:
"""The worksheet PARTS the workbook marks hidden.
The sheet file says nothing about it: the state lives in `workbook.xml`
and the part is reached through the relationship id."""
names = archive.namelist()
if "xl/workbook.xml" not in names or "xl/_rels/workbook.xml.rels" not in names:
return set()
relationships = ET.fromstring(archive.read("xl/_rels/workbook.xml.rels"))
targets = {
node.get("Id", ""): str(node.get("Target", ""))
for node in relationships
if _local(node.tag) == "Relationship"
}
hidden: set[str] = set()
workbook = ET.fromstring(archive.read("xl/workbook.xml"))
for sheet in workbook.iter(f"{_S}sheet"):
if sheet.get("state") in ("hidden", "veryHidden"):
rid = next((v for k, v in sheet.attrib.items() if _local(k) == "id"), "")
target = targets.get(rid, "")
if target:
hidden.add(f"xl/{target.lstrip('/')}" if not target.startswith("xl/") else target)
return hidden
def count_xlsx(data: bytes) -> Count:
"""sheet: xl/worksheets/sheetN.xml. row: a row holding a value. cell: a c
with a value. image: an xdr:pic in a drawing.
A cell's text is resolved through `sharedStrings.xml`, because that is
where a spreadsheet's words actually live: the cell holds an index."""
count = Count(XLSX)
with zipfile.ZipFile(io.BytesIO(data)) as archive:
shared = _shared_strings(archive)
hidden = _hidden_sheets(archive)
for name in sorted(archive.namelist()):
if re.fullmatch(r"xl/worksheets/sheet\d+\.xml", name):
root = ET.fromstring(archive.read(name))
sheet_pieces: list[str] = []
for row in root.iter(f"{_S}row"):
valued = [
c
for c in row.iter(f"{_S}c")
if c.find(f"{_S}v") is not None or c.find(f"{_S}is") is not None
]
values = [_cell_text(c, shared) for c in valued]
for value in values:
count.add("cell", value)
for cell in valued:
formula = cell.find(f"{_S}f")
if formula is not None:
count.add("formula", _normal(formula.text or ""))
if valued:
count.add("row", *values)
sheet_pieces.extend(values)
count.add("hidden_sheet" if name in hidden else "sheet", *sheet_pieces)
elif re.fullmatch(r"xl/drawings/drawing\d+\.xml", name):
root = ET.fromstring(archive.read(name))
for _ in root.iter(f"{_XDR}pic"):
count.add("image")
return count
ODT = (
"annotation",
"cell",
"header_footer",
"heading",
"image",
"list_item",
"paragraph",
"table",
)
_OFFICE = "{urn:oasis:names:tc:opendocument:xmlns:office:1.0}"
_STYLE = "{urn:oasis:names:tc:opendocument:xmlns:style:1.0}"
def count_odt(data: bytes) -> Count:
"""heading: text:h. paragraph: a text:p with text outside a table cell.
table: table:table. cell: table:table-cell. list_item: text:list-item.
image: draw:image."""
styles = None
with zipfile.ZipFile(io.BytesIO(data)) as archive:
root = ET.fromstring(archive.read("content.xml"))
if "styles.xml" in archive.namelist():
styles = ET.fromstring(archive.read("styles.xml"))
count = Count(ODT)
in_cell: set[int] = set()
for cell in root.iter(f"{_TABLE}table-cell"):
in_cell.update(id(p) for p in cell.iter(f"{_TEXT}p"))
# A comment is not prose. Counted as a paragraph it makes the accounting
# demand that a reader carry a note the author wrote to themselves.
annotated: set[int] = set()
for note in root.iter(f"{_OFFICE}annotation"):
annotated.update(id(p) for p in note.iter(f"{_TEXT}p"))
count.add(
"annotation",
*[_normal("".join(p.itertext())) for p in note.iter(f"{_TEXT}p")],
)
def lines(node: ET.Element) -> list[str]:
pieces = [_normal("".join(p.itertext())) for p in node.iter(f"{_TEXT}p")]
pieces += [_normal("".join(h.itertext())) for h in node.iter(f"{_TEXT}h")]
return [piece for piece in pieces if piece]
for heading in root.iter(f"{_TEXT}h"):
count.add("heading", _normal("".join(heading.itertext())))
for para in root.iter(f"{_TEXT}p"):
text = _normal("".join(para.itertext()))
if id(para) not in in_cell and id(para) not in annotated and text:
count.add("paragraph", text)
for table in root.iter(f"{_TABLE}table"):
count.add("table", *lines(table))
for cell in root.iter(f"{_TABLE}table-cell"):
count.add("cell", *lines(cell))
for item in root.iter(f"{_TEXT}list-item"):
count.add("list_item", *lines(item))
for _ in root.iter(f"{_DRAW}image"):
count.add("image")
# The header and the footer live in `styles.xml`, which is why no reader
# looking only at `content.xml` can see them at all.
if styles is not None:
for place in (f"{_STYLE}header", f"{_STYLE}footer"):
for region in styles.iter(place):
for para in region.iter(f"{_TEXT}p"):
text = _normal("".join(para.itertext()))
if text:
count.add("header_footer", text)
return count
RTF = ("cell", "image", "paragraph", "table_row")
_RTF_CONTROL = re.compile(r"\\([a-zA-Z]+)(-?\d+)? ?|\\'([0-9a-fA-F]{2})|\\([^a-zA-Z])")
#: Groups that hold no document text. `{\*...}` says so in the format itself;
#: these say it by name, and without them a fixture's font table reads as the
#: first paragraph of its prose and a picture's hex payload as the next one.
_RTF_SILENT = frozenset(
{
"fonttbl",
"pict",
"colortbl",
"stylesheet",
"info",
"listtable",
"listoverridetable",
"rsidtbl",
"generator",
"filetbl",
"pgptbl",
"themedata",
"colorschememapping",
"latentstyles",
"datastore",
}
)
def _rtf_pieces(text: str) -> dict[str, list[str]]:
r"""The text standing before each `\par`, `\cell` and `\row`.
A control word ends at the first non-letter, so `\pard` is not `\par`.
`\uN` carries a character the byte escapes cannot, and the substitution
character standing after it is the same character again -- counted twice,
a Norwegian word reads as `hovedl?pet`.
"""
pieces: dict[str, list[Any]] = {"paragraph": [], "cell": [], "table_row": []}
buffer: list[str] = []
cell: list[str] = []
silent: list[int] = []
depth = 0
position = 0
skip_after_unicode = 0
while position < len(text):
char = text[position]
quiet = bool(silent)
if char == "{":
depth += 1
position += 1
match = re.match(r"\\\*?\\?([a-zA-Z]+)", text[position : position + 32])
if match and match.group(1) in _RTF_SILENT:
silent.append(depth)
elif text[position : position + 2] == "\\*":
silent.append(depth)
continue
if char == "}":
if silent and silent[-1] == depth:
silent.pop()
depth -= 1
position += 1
continue
if char == "\\":
match = _RTF_CONTROL.match(text, position)
if match is None:
position += 1
continue
position = match.end()
word, number, hexcode, symbol = match.groups()
if hexcode is not None:
if not quiet and not skip_after_unicode:
buffer.append(bytes([int(hexcode, 16)]).decode("cp1252", "replace"))
skip_after_unicode = max(0, skip_after_unicode - 1)
continue
if symbol is not None:
continue
if word == "u" and number is not None:
code = int(number)
if not quiet:
buffer.append(chr(code if code >= 0 else code + 65536))
skip_after_unicode = 1
continue
if word == "par":
pieces["paragraph"].append(_normal("".join(buffer)))
buffer = []
elif word == "cell":
joined = _normal("".join(buffer))
pieces["cell"].append(joined)
cell.append(joined)
buffer = []
elif word == "row":
pieces["table_row"].append(cell)
cell = []
continue
if skip_after_unicode and not char.isspace():
skip_after_unicode -= 1
position += 1
continue
if not quiet:
buffer.append(char)
position += 1
return pieces
def count_rtf(text: str) -> Count:
"""paragraph: \\par. table_row: \\row. cell: \\cell. image: \\pict."""
count = Count(RTF)
def word(name: str) -> int:
return len(re.findall(rf"\\{name}(?![a-zA-Z])", text))
pieces = _rtf_pieces(text)
for role, control in (("paragraph", "par"), ("cell", "cell"), ("table_row", "row")):
found = pieces[role]
for index in range(word(control)):
own = found[index] if index < len(found) else ""
count.add(role, *(own if isinstance(own, list) else [own]))
for _ in range(word("pict")):
count.add("image")
return count
# --- pdf ---------------------------------------------------------------------
PDF = ("image", "page")
def _lines(text: str) -> list[str]:
"""One piece per line of a page: a reader re-wraps, and a line survives."""
return [_normal(line) for line in text.split("\n") if line.strip()]
def pdf_objects(path: Path) -> Count | None:
"""page and image placements as pdfplumber sees them; None without it.
A page's text is the page's own text. Until 2026-09-18 this witness saw
a PDF as pages and picture placements alone, so ALL of a PDF's text could
leave the bundle with no row able to see it (independent review, M-1).
"""
try:
import pdfplumber
except ImportError:
return None
count = Count(PDF)
with pdfplumber.open(str(path)) as pdf:
for page in pdf.pages:
for _ in page.images:
count.add("image")
count.add("page", *_lines(page.extract_text() or ""))
page.close()
return count
def pdf_poppler(path: Path) -> Count | None:
"""The same, through poppler (`pdfinfo`, `pdfimages -list`, `pdftotext`);
None when poppler is not installed.
A genuinely independent reader: a different code base, a different text
engine, run as a subprocess."""
info = shutil.which("pdfinfo")
lister = shutil.which("pdfimages")
totext = shutil.which("pdftotext")
if info is None or lister is None or totext is None:
return None
meta = subprocess.run([info, str(path)], capture_output=True, text=True, check=True).stdout
pages_match = re.search(r"^Pages:\s+(\d+)", meta, re.MULTILINE)
pages = int(pages_match.group(1)) if pages_match else 0
listing = subprocess.run(
[lister, "-list", str(path)], capture_output=True, text=True, check=True
).stdout
count = Count(PDF)
for line in listing.splitlines()[2:]:
cells = line.split()
if len(cells) > 2 and cells[2] == "image":
count.add("image")
for number in range(1, pages + 1):
page = subprocess.run(
[totext, "-f", str(number), "-l", str(number), str(path), "-"],
capture_output=True,
text=True,
check=True,
).stdout
count.add("page", *_lines(page))
return count
# --- one file ----------------------------------------------------------------
WITNESSED_SUFFIXES = (
".csv",
".docx",
".htm",
".html",
".json",
".md",
".odt",
".pdf",
".pptx",
".rtf",
".txt",
".xlsx",
".xml",
)
#: What each witness STILL does not count, by name and per file type.
#:
#: An accounting can only lose visibly what something counts, so this list is
#: the gate's own statement of its blind spots -- printed on every run, never
#: inferred, and the raw material for the next capability order. Written
#: 2026-09-18 from an independent review's per-format reading of this file.
NOT_COUNTED: dict[str, tuple[str, ...]] = {
".csv": (
"a semicolon-separated file (the Norwegian default) reads as one cell per row",
"quoting and encoding errors, which arrive as text",
),
".docx": (
"SmartArt, charts and embedded OLE objects",
"tracked deletions",
"hyperlink targets (the link text counts, the address does not)",
"an equation written as `m:oMath` (it holds `m:t`, not `w:t`)",
"a picture's alt text",
),
".htm": (
"text in `div`, `blockquote`, `pre`, `dd`, `figcaption`, `caption` and bare text",
"`alt` and `title` attributes",
"`details`/`summary`, `picture`/`source`, `svg`, `object`, `iframe`",
),
".html": (
"text in `div`, `blockquote`, `pre`, `dd`, `figcaption`, `caption` and bare text",
"`alt` and `title` attributes",
"`details`/`summary`, `picture`/`source`, `svg`, `object`, `iframe`",
),
".json": ("the order of members, and comments a JSON superset would allow",),
".md": (
"Setext headings (`===`, `---`)",
"reference images `![a][r]` and raw `<img>`",
"footnotes, indented code blocks and front matter (they count as paragraphs)",
),
".odt": (
"tracked changes",
"`draw:object` (an embedded chart or formula)",
"a picture's `xlink:href`: an odt image counts as embedded, so its bytes "
"cannot be traced to an inbox file",
),
".pdf": (
"headings, paragraphs and tables as such (operator-approved exception, "
"2026-09-17): the page's text is counted, its structure is not",
"form fields, annotations, attachments and bookmarks",
),
".pptx": (
"SmartArt, charts and comments",
"alt text",
"a slide layout's and master's own text",
),
".rtf": (
"text inside `\\header`, `\\footer` and `\\footnote` counts as body prose",
"a picture Word writes twice (`\\shppict` and `\\nonshppict`) counts twice",
"a picture's payload is binary, so the paragraph holding it has no text "
"the gate can look for",
"an empty `\\par` counts as a paragraph, where docx counts only one with text",
),
".txt": ("nothing beyond lines and paragraphs: the format declares no more",),
".xlsx": (
"merged cells, cell comments, defined names and charts",
"a sheet's NAME",
"number formats (a date reads as its serial number)",
),
".xml": (
"`ref` and `element-citation` outside `mixed-citation`",
"`def-list`, `term-sec` and `app`",
"a `non-normative-note`'s label",
"attributes, and all text of an XML document that is not NISO-STS beyond "
"the element's own text",
),
}
def witness_file(inbox: Path, path: Path) -> Inventory:
"""Count one file under `inbox`. Raises WitnessRefused for a file the
witness does not read."""
suffix = path.suffix.lower()
relative = path.relative_to(inbox).as_posix()
data = path.read_bytes()
refs: list[str] = []
sts = False
if suffix == ".md":
count, refs = count_markdown(data.decode("utf-8-sig"))
name = "markdown lines"
elif suffix == ".txt":
count, name = count_text(data.decode("utf-8-sig")), "text lines"
elif suffix == ".csv":
count, name = count_csv(data.decode("utf-8-sig")), "csv"
elif suffix == ".json":
if is_sts_json(data):
count, name = count_sts_json(data), "sts json node tree"
else:
count, name = count_json(data.decode("utf-8-sig")), "json"
elif suffix in (".html", ".htm"):
parser = _HtmlCounter()
parser.feed(data.decode("utf-8-sig"))
parser.close()
parser.finish()
count, refs, name = parser.count, parser.refs, "html.parser"
elif suffix == ".xml":
count, refs, sts = count_sts_xml(data)
name = "xml.etree"
elif suffix == ".docx":
count, name = count_docx(data), "docx zip xml"
elif suffix == ".pptx":
count, name = count_pptx(data), "pptx zip xml"
elif suffix == ".xlsx":
count, name = count_xlsx(data), "xlsx zip xml"
elif suffix == ".odt":
count, name = count_odt(data), "odt zip xml"
elif suffix == ".rtf":
count, name = count_rtf(data.decode("latin-1")), "rtf control words"
elif suffix == ".pdf":
objects = pdf_objects(path)
if objects is None:
raise WitnessRefused("pdfplumber is not installed")
count, name = objects, "pdfplumber objects"
else:
raise WitnessRefused(f"no witness for {suffix or 'a file without a suffix'}")
inventory = Inventory(
relative,
suffix,
name,
dict(count.counts),
texts={role: [list(p) for p in values] for role, values in count.texts.items()},
)
for ref in refs:
if not ref or _is_remote(ref):
inventory.images.append(ImageRef(REMOTE, ref))
else:
target = resolve_local(inbox, path, ref, sts=sts)
inventory.images.append(ImageRef(LOCAL, ref, target))
embedded = count.counts.get("image", 0) - len(refs)
inventory.images.extend(ImageRef(EMBEDDED, "") for _ in range(max(0, embedded)))
return inventory
def walk(inbox: Path) -> Iterator[Path]:
"""Every file under `inbox`, sorted by relative path, dot-entries skipped."""
for path in sorted(inbox.rglob("*"), key=lambda p: p.relative_to(inbox).as_posix()):
if path.is_file() and not any(
part.startswith(".") for part in path.relative_to(inbox).parts
):
yield path
def witness_inbox(inbox: Path) -> dict[str, object]:
"""The committed fasit form: every document's inventory, every other file
with the documents that point at it."""
documents: dict[str, object] = {}
others: list[str] = []
pointed: dict[str, list[str]] = {}
for path in walk(inbox):
if path.suffix.lower() not in WITNESSED_SUFFIXES:
others.append(path.relative_to(inbox).as_posix())
continue
inventory = witness_file(inbox, path)
documents[inventory.source_file] = inventory.to_json()
for ref in inventory.images:
if ref.target is not None:
pointed.setdefault(ref.target, [])
if inventory.source_file not in pointed[ref.target]:
pointed[ref.target].append(inventory.source_file)
return {
"witness_version": WITNESS_VERSION,
"documents": documents,
"files": {name: {"pointed_at_by": pointed.get(name, [])} for name in others},
}
def main(argv: list[str] | None = None) -> int:
import argparse
parser = argparse.ArgumentParser(description=__doc__.split("\n", 1)[0])
parser.add_argument("inbox", type=Path)
args = parser.parse_args(argv)
print(json.dumps(witness_inbox(args.inbox), indent=2, ensure_ascii=False, sort_keys=True))
return 0
if __name__ == "__main__":
raise SystemExit(main())