Capability loop step 3, no capability. tools/okf_accounting_gate.py asks, per README file type, how many of the elements a SOURCE holds okf build books as carried / pointer / coded rejection, with unaccounted and double-booked both required to be 0. Exit 1 today on rows 2, 3, 4 and 6. The fasit is tools/okf_witness.py (stdlib + pdfplumber + poppler, no package import; tested on the live import graph), committed as tests/fixtures/accounting/*inventory.json over one fixture per type. Measured: no source inventory (0 of 13); two graphics/ files carried through documents AND counted extractor_unknown (50 on R761 under --gate none); a refused document logged "0 carried of 0 found"; R761 refused whole because guard 1.4.0 treats its 71 U+00AD soft hyphens as an invisible carrier (asked of the security repo). The two R761 witnesses agree once STS labels are counted by role, not tag. Report: docs/2026-09-17-innholdsregnskapet-rod-gate.md Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
697 lines
25 KiB
Python
697 lines
25 KiB
Python
"""The independent witness: what a SOURCE file holds, counted by the format's rules.
|
|
|
|
This module is the fasit side of the content-accounting gate
|
|
(`tools/okf_accounting_gate.py`). It answers one question per file -- "how many
|
|
of each element does this source carry?" -- and it answers it WITHOUT this
|
|
package: no `llm_ingestion_okf` module is imported, directly or through a
|
|
helper, and `tests/test_accounting_gate.py` proves that on the live import
|
|
graph rather than by searching the text. A fasit computed by the reader it is
|
|
meant to judge is the reader agreeing with itself.
|
|
|
|
Each counter reads the container the way the format defines it:
|
|
|
|
- XML (NISO-STS): `xml.etree.ElementTree` straight on the bytes.
|
|
- STS JSON twin: the publisher's own node tree (`standardContent`, nodes with
|
|
`e`/`t`/`x`), walked with the same element roles as the XML.
|
|
- docx / pptx / xlsx / odt: the zip members' own XML.
|
|
- PDF: pdfplumber OBJECTS (pages, image placements) and, as a second witness,
|
|
poppler (`pdfinfo`, `pdfimages -list`). pdfplumber is also what the reader
|
|
extracts text with, which is why the gate never trusts a PDF count that the
|
|
poppler side does not repeat.
|
|
- HTML: `html.parser` from the stdlib.
|
|
- md / txt / csv / json / rtf: stdlib line, csv and json readers, and a
|
|
control-word scan for rtf.
|
|
|
|
The ELEMENT VOCABULARY is part of the gate's contract: a build that declares
|
|
an inventory must use these names, per file type, or it is not comparable.
|
|
Each name is defined where it is counted, and a type counts only what that
|
|
format actually carries.
|
|
|
|
Image REFERENCES are resolved here too, because the accounting has to know
|
|
which inbox files a document points at: a relative reference is taken against
|
|
the document's own directory, never above it, and an STS reference that is
|
|
not found there is looked for as `graphics/<basename>` -- the layout the
|
|
publisher's STS delivery ships in. That is a fact about the delivery format,
|
|
written down here, not borrowed from the reader.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import io
|
|
import json
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import zipfile
|
|
from collections.abc import Iterator
|
|
from dataclasses import dataclass, field
|
|
from html.parser import HTMLParser
|
|
from pathlib import Path, PurePosixPath
|
|
from xml.etree import ElementTree as ET
|
|
|
|
WITNESS_VERSION = 1
|
|
|
|
#: Pointer kinds a document can hold for an image.
|
|
LOCAL = "local"
|
|
REMOTE = "remote"
|
|
EMBEDDED = "embedded"
|
|
|
|
|
|
class WitnessRefused(Exception):
|
|
"""The witness will not read this file (for example a DOCTYPE)."""
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ImageRef:
|
|
"""One image a document declares.
|
|
|
|
`target` is the inbox-relative POSIX path of the file a LOCAL reference
|
|
resolves to, or None when it resolves to nothing inside the document's
|
|
directory.
|
|
"""
|
|
|
|
kind: str
|
|
ref: str
|
|
target: str | None = None
|
|
|
|
|
|
@dataclass
|
|
class Inventory:
|
|
"""What one source file holds, element type by element type."""
|
|
|
|
source_file: str
|
|
suffix: str
|
|
witness: str
|
|
elements: dict[str, int] = field(default_factory=dict)
|
|
images: list[ImageRef] = field(default_factory=list)
|
|
|
|
@property
|
|
def total(self) -> int:
|
|
return sum(self.elements.values())
|
|
|
|
def to_json(self) -> dict[str, object]:
|
|
return {
|
|
"suffix": self.suffix,
|
|
"witness": self.witness,
|
|
"elements": dict(sorted(self.elements.items())),
|
|
"images": [
|
|
{"kind": ref.kind, "ref": ref.ref, "target": ref.target} for ref in self.images
|
|
],
|
|
}
|
|
|
|
|
|
def _local(tag: str) -> str:
|
|
return tag.rsplit("}", 1)[-1] if "}" in tag else tag.split(":")[-1]
|
|
|
|
|
|
def _is_remote(ref: str) -> bool:
|
|
return bool(re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*:", ref)) or ref.startswith("//")
|
|
|
|
|
|
def resolve_local(inbox: Path, document: Path, ref: str, *, sts: bool = False) -> str | None:
|
|
"""The inbox-relative path a LOCAL reference names, or None.
|
|
|
|
Contained in the document's own directory: an absolute path or one that
|
|
climbs above that directory resolves to nothing.
|
|
"""
|
|
base = document.parent
|
|
candidates = [ref]
|
|
if sts:
|
|
candidates.append(f"graphics/{PurePosixPath(ref).name}")
|
|
for candidate in candidates:
|
|
pure = PurePosixPath(candidate)
|
|
if pure.is_absolute() or ".." in pure.parts:
|
|
continue
|
|
target = base.joinpath(*pure.parts)
|
|
if target.is_file():
|
|
return target.relative_to(inbox).as_posix()
|
|
return None
|
|
|
|
|
|
# --- markdown / text ---------------------------------------------------------
|
|
|
|
_FENCE_OPEN = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$")
|
|
_ATX = re.compile(r"^ {0,3}#{1,6}(\s|$)")
|
|
_DELIMITER_ROW = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
|
|
_MD_IMAGE = re.compile(r"!\[[^\]]*\]\(\s*<?([^)\s>]+)>?[^)]*\)")
|
|
|
|
|
|
def _unfenced(lines: list[str]) -> tuple[list[str | None], int]:
|
|
"""Lines with fenced ones replaced by None, and the number of fences.
|
|
|
|
CommonMark SS 4.5 in the parts that decide which lines are fenced: up to
|
|
three leading spaces, a backtick info string may not hold a backtick, the
|
|
closing fence is the same character and at least as long, and an unclosed
|
|
fence runs to the end of the text.
|
|
"""
|
|
out: list[str | None] = []
|
|
fences = 0
|
|
opener: str | None = None
|
|
for line in lines:
|
|
if opener is None:
|
|
match = _FENCE_OPEN.match(line)
|
|
if match and not (match.group(1)[0] == "`" and "`" in match.group(2)):
|
|
opener = match.group(1)
|
|
fences += 1
|
|
out.append(None)
|
|
continue
|
|
out.append(line)
|
|
continue
|
|
out.append(None)
|
|
stripped = line.strip()
|
|
if (
|
|
stripped
|
|
and set(stripped) == {opener[0]}
|
|
and len(stripped) >= len(opener)
|
|
and len(line) - len(line.lstrip(" ")) <= 3
|
|
):
|
|
opener = None
|
|
return out, fences
|
|
|
|
|
|
def count_markdown(text: str) -> tuple[dict[str, int], list[str]]:
|
|
"""heading: ATX lines outside a fence. table: a pipe row followed by a
|
|
delimiter row. table_row: the body rows under it. image: ``
|
|
outside a fence. code_block: a fence. paragraph: a run of non-blank lines
|
|
outside a fence that holds none of the above."""
|
|
lines, fences = _unfenced(text.split("\n"))
|
|
elements = {
|
|
"heading": 0,
|
|
"paragraph": 0,
|
|
"table": 0,
|
|
"table_row": 0,
|
|
"image": 0,
|
|
"code_block": fences,
|
|
}
|
|
refs: list[str] = []
|
|
in_table = False
|
|
in_paragraph = False
|
|
delimiter_rows: set[int] = set()
|
|
for index, line in enumerate(lines):
|
|
if index in delimiter_rows:
|
|
continue
|
|
if line is None or not line.strip():
|
|
in_table = False
|
|
in_paragraph = False
|
|
continue
|
|
if in_table:
|
|
if "|" in line:
|
|
elements["table_row"] += 1
|
|
continue
|
|
in_table = False
|
|
following = lines[index + 1] if index + 1 < len(lines) else None
|
|
if "|" in line and following is not None and _DELIMITER_ROW.match(following):
|
|
elements["table"] += 1
|
|
in_table = True
|
|
delimiter_rows.add(index + 1) # the delimiter row is not a body row
|
|
in_paragraph = False
|
|
continue
|
|
if _ATX.match(line):
|
|
elements["heading"] += 1
|
|
in_paragraph = False
|
|
continue
|
|
found = _MD_IMAGE.findall(line)
|
|
if found:
|
|
elements["image"] += len(found)
|
|
refs.extend(found)
|
|
if not _MD_IMAGE.sub("", line).strip():
|
|
in_paragraph = False
|
|
continue
|
|
if not in_paragraph:
|
|
elements["paragraph"] += 1
|
|
in_paragraph = True
|
|
return elements, refs
|
|
|
|
|
|
def count_text(text: str) -> dict[str, int]:
|
|
"""paragraph: a run of non-blank lines. line: a non-blank line."""
|
|
paragraphs = 0
|
|
lines = 0
|
|
previous_blank = True
|
|
for line in text.split("\n"):
|
|
if line.strip():
|
|
lines += 1
|
|
if previous_blank:
|
|
paragraphs += 1
|
|
previous_blank = False
|
|
else:
|
|
previous_blank = True
|
|
return {"paragraph": paragraphs, "line": lines}
|
|
|
|
|
|
def count_csv(text: str) -> dict[str, int]:
|
|
"""header_cell: cells of the first row. row / cell: every row after it."""
|
|
rows = [row for row in csv.reader(io.StringIO(text)) if row]
|
|
if not rows:
|
|
return {"header_cell": 0, "row": 0, "cell": 0}
|
|
return {
|
|
"header_cell": len(rows[0]),
|
|
"row": len(rows) - 1,
|
|
"cell": sum(len(row) for row in rows[1:]),
|
|
}
|
|
|
|
|
|
def count_json(text: str) -> dict[str, int]:
|
|
"""key: an object member. value: a leaf (string, number, boolean, null)."""
|
|
counts = {"key": 0, "value": 0}
|
|
|
|
def walk(node: object) -> None:
|
|
if isinstance(node, dict):
|
|
counts["key"] += len(node)
|
|
for child in node.values():
|
|
walk(child)
|
|
elif isinstance(node, list):
|
|
for child in node:
|
|
walk(child)
|
|
else:
|
|
counts["value"] += 1
|
|
|
|
walk(json.loads(text))
|
|
return counts
|
|
|
|
|
|
# --- html --------------------------------------------------------------------
|
|
|
|
|
|
class _HtmlCounter(HTMLParser):
|
|
"""heading: h1-h6. paragraph: p. list_item: li. table: table. cell: td,
|
|
th. image: img."""
|
|
|
|
_ROLES = {
|
|
**{f"h{level}": "heading" for level in range(1, 7)},
|
|
"p": "paragraph",
|
|
"li": "list_item",
|
|
"table": "table",
|
|
"td": "cell",
|
|
"th": "cell",
|
|
"img": "image",
|
|
}
|
|
|
|
def __init__(self) -> None:
|
|
super().__init__(convert_charrefs=True)
|
|
self.elements = {role: 0 for role in sorted(set(self._ROLES.values()))}
|
|
self.refs: list[str] = []
|
|
|
|
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
role = self._ROLES.get(tag)
|
|
if role is None:
|
|
return
|
|
self.elements[role] += 1
|
|
if tag == "img":
|
|
self.refs.append(dict(attrs).get("src") or "")
|
|
|
|
|
|
# --- xml / sts ---------------------------------------------------------------
|
|
|
|
STS_ROLES = (
|
|
"section",
|
|
"title",
|
|
"section_label",
|
|
"paragraph",
|
|
"table",
|
|
"table_label",
|
|
"cell",
|
|
"list_item",
|
|
"image",
|
|
"footnote",
|
|
)
|
|
|
|
|
|
def _sts_role(tag: str, parent: str | None, grandparent: str | None) -> str | None:
|
|
"""The one mapping from an STS element to its accounting role.
|
|
|
|
Used by BOTH STS witnesses, and the role is the unit, not the tag, because
|
|
the publisher's two deliveries of one document place the same text
|
|
differently (measured on R761 Prosesskoden:2025, 2026-09-17):
|
|
|
|
- a section's label: XML `sec/label` on 7 714 sections; JSON `sec/label`
|
|
on 4 954 and `sec/title/label` on the 2 760 that carry a title.
|
|
- a table's label: XML `table-wrap/label` (10); JSON
|
|
`table-wrap/table/caption` (10).
|
|
|
|
Counted by tag, the two witnesses disagree by 2 760 and by 10 on text
|
|
both of them carry.
|
|
"""
|
|
if tag == "sec":
|
|
return "section"
|
|
if tag == "title" and parent == "sec":
|
|
return "title"
|
|
if tag == "label" and (parent == "sec" or (parent == "title" and grandparent == "sec")):
|
|
return "section_label"
|
|
if tag == "label" and parent == "table-wrap":
|
|
return "table_label"
|
|
if tag == "caption" and parent == "table" and grandparent == "table-wrap":
|
|
return "table_label"
|
|
if tag == "p":
|
|
return "paragraph"
|
|
if tag == "table-wrap":
|
|
return "table"
|
|
if tag in ("td", "th"):
|
|
return "cell"
|
|
if tag == "list-item":
|
|
return "list_item"
|
|
if tag in ("graphic", "inline-graphic"):
|
|
return "image"
|
|
if tag == "fn":
|
|
return "footnote"
|
|
return None
|
|
|
|
|
|
def count_sts_xml(data: bytes) -> tuple[dict[str, int], list[str], bool]:
|
|
"""Element roles of an STS document; `element` alone for other XML."""
|
|
if b"<!DOCTYPE" in data:
|
|
raise WitnessRefused("a DOCTYPE is not parsed")
|
|
root = ET.fromstring(data)
|
|
sts = _local(root.tag) == "standard" or any(_local(el.tag) == "sec" for el in root.iter())
|
|
if not sts:
|
|
return {"element": sum(1 for _ in root.iter())}, [], False
|
|
elements = {role: 0 for role in STS_ROLES}
|
|
refs: list[str] = []
|
|
|
|
def walk(node: ET.Element, parent: str | None, grandparent: str | None) -> None:
|
|
tag = _local(node.tag)
|
|
role = _sts_role(tag, parent, grandparent)
|
|
if role is not None:
|
|
elements[role] += 1
|
|
if role == "image":
|
|
href = next((value for key, value in node.attrib.items() if _local(key) == "href"), "")
|
|
refs.append(href)
|
|
for child in node:
|
|
walk(child, tag, parent)
|
|
|
|
walk(root, None, None)
|
|
return elements, refs, True
|
|
|
|
|
|
def count_sts_json(data: bytes) -> dict[str, int]:
|
|
"""The same roles, read from the publisher's JSON node tree."""
|
|
document = json.loads(data)
|
|
elements = {role: 0 for role in STS_ROLES}
|
|
|
|
def walk(node: dict[str, object], parent: str | None, grandparent: str | None) -> None:
|
|
body = node.get("x")
|
|
if not isinstance(body, dict):
|
|
return
|
|
tag = str(body.get("tag"))
|
|
role = _sts_role(tag, parent, grandparent)
|
|
if role is not None:
|
|
elements[role] += 1
|
|
for child in body.get("c") or []:
|
|
walk(child, tag, parent)
|
|
|
|
for child in document["standardContent"]["c"]:
|
|
walk(child, None, None)
|
|
return elements
|
|
|
|
|
|
# --- office zips -------------------------------------------------------------
|
|
|
|
_W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
|
|
_A = "{http://schemas.openxmlformats.org/drawingml/2006/main}"
|
|
_P = "{http://schemas.openxmlformats.org/presentationml/2006/main}"
|
|
_S = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
|
|
_XDR = "{http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing}"
|
|
_TEXT = "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}"
|
|
_TABLE = "{urn:oasis:names:tc:opendocument:xmlns:table:1.0}"
|
|
_DRAW = "{urn:oasis:names:tc:opendocument:xmlns:drawing:1.0}"
|
|
_HEADING_STYLE = re.compile(r"^(heading|overskrift|title|tittel)\s*\d*$", re.IGNORECASE)
|
|
|
|
|
|
def _text_of(node: ET.Element, tag: str) -> str:
|
|
return "".join(t.text or "" for t in node.iter(tag))
|
|
|
|
|
|
def count_docx(data: bytes) -> dict[str, int]:
|
|
"""heading: a w:p whose style is a heading or title style. paragraph: any
|
|
other w:p with text. table: w:tbl. cell: w:tc. image: a:blip.
|
|
footnote: a w:footnote with a positive id."""
|
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
root = ET.fromstring(archive.read("word/document.xml"))
|
|
footnotes = 0
|
|
if "word/footnotes.xml" in archive.namelist():
|
|
notes = ET.fromstring(archive.read("word/footnotes.xml"))
|
|
footnotes = sum(
|
|
1 for note in notes.iter(f"{_W}footnote") if int(note.get(f"{_W}id", "0")) > 0
|
|
)
|
|
headings = paragraphs = 0
|
|
for para in root.iter(f"{_W}p"):
|
|
style = para.find(f"{_W}pPr/{_W}pStyle")
|
|
if style is not None and _HEADING_STYLE.match(style.get(f"{_W}val", "")):
|
|
headings += 1
|
|
elif _text_of(para, f"{_W}t").strip():
|
|
paragraphs += 1
|
|
return {
|
|
"heading": headings,
|
|
"paragraph": paragraphs,
|
|
"table": sum(1 for _ in root.iter(f"{_W}tbl")),
|
|
"cell": sum(1 for _ in root.iter(f"{_W}tc")),
|
|
"image": sum(1 for _ in root.iter(f"{_A}blip")),
|
|
"footnote": footnotes,
|
|
}
|
|
|
|
|
|
def count_pptx(data: bytes) -> dict[str, int]:
|
|
"""slide: ppt/slides/slideN.xml. title: a shape whose placeholder is a
|
|
title. paragraph: an a:p with text outside a table and outside a title.
|
|
table: a:tbl. cell: a:tc. image: p:pic."""
|
|
counts = {"slide": 0, "title": 0, "paragraph": 0, "table": 0, "cell": 0, "image": 0}
|
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
slides = [n for n in archive.namelist() if re.fullmatch(r"ppt/slides/slide\d+\.xml", n)]
|
|
for name in slides:
|
|
counts["slide"] += 1
|
|
root = ET.fromstring(archive.read(name))
|
|
counts["table"] += sum(1 for _ in root.iter(f"{_A}tbl"))
|
|
counts["cell"] += sum(1 for _ in root.iter(f"{_A}tc"))
|
|
counts["image"] += sum(1 for _ in root.iter(f"{_P}pic"))
|
|
for shape in root.iter(f"{_P}sp"):
|
|
placeholder = shape.find(f"{_P}nvSpPr/{_P}nvPr/{_P}ph")
|
|
is_title = placeholder is not None and placeholder.get("type") in (
|
|
"title",
|
|
"ctrTitle",
|
|
)
|
|
texts = [p for p in shape.iter(f"{_A}p") if _text_of(p, f"{_A}t").strip()]
|
|
if is_title and texts:
|
|
counts["title"] += 1
|
|
else:
|
|
counts["paragraph"] += len(texts)
|
|
return counts
|
|
|
|
|
|
def count_xlsx(data: bytes) -> dict[str, int]:
|
|
"""sheet: xl/worksheets/sheetN.xml. row: a row holding a value. cell: a c
|
|
with a value. image: an xdr:pic in a drawing."""
|
|
counts = {"sheet": 0, "row": 0, "cell": 0, "image": 0}
|
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
for name in archive.namelist():
|
|
if re.fullmatch(r"xl/worksheets/sheet\d+\.xml", name):
|
|
counts["sheet"] += 1
|
|
root = ET.fromstring(archive.read(name))
|
|
for row in root.iter(f"{_S}row"):
|
|
valued = [
|
|
c
|
|
for c in row.iter(f"{_S}c")
|
|
if c.find(f"{_S}v") is not None or c.find(f"{_S}is") is not None
|
|
]
|
|
counts["cell"] += len(valued)
|
|
counts["row"] += 1 if valued else 0
|
|
elif re.fullmatch(r"xl/drawings/drawing\d+\.xml", name):
|
|
root = ET.fromstring(archive.read(name))
|
|
counts["image"] += sum(1 for _ in root.iter(f"{_XDR}pic"))
|
|
return counts
|
|
|
|
|
|
def count_odt(data: bytes) -> dict[str, int]:
|
|
"""heading: text:h. paragraph: a text:p with text outside a table cell.
|
|
table: table:table. cell: table:table-cell. list_item: text:list-item.
|
|
image: draw:image."""
|
|
with zipfile.ZipFile(io.BytesIO(data)) as archive:
|
|
root = ET.fromstring(archive.read("content.xml"))
|
|
in_cell: set[int] = set()
|
|
for cell in root.iter(f"{_TABLE}table-cell"):
|
|
in_cell.update(id(p) for p in cell.iter(f"{_TEXT}p"))
|
|
return {
|
|
"heading": sum(1 for _ in root.iter(f"{_TEXT}h")),
|
|
"paragraph": sum(
|
|
1
|
|
for p in root.iter(f"{_TEXT}p")
|
|
if id(p) not in in_cell and "".join(p.itertext()).strip()
|
|
),
|
|
"table": sum(1 for _ in root.iter(f"{_TABLE}table")),
|
|
"cell": sum(1 for _ in root.iter(f"{_TABLE}table-cell")),
|
|
"list_item": sum(1 for _ in root.iter(f"{_TEXT}list-item")),
|
|
"image": sum(1 for _ in root.iter(f"{_DRAW}image")),
|
|
}
|
|
|
|
|
|
def count_rtf(text: str) -> dict[str, int]:
|
|
"""paragraph: \\par. table_row: \\row. cell: \\cell. image: \\pict.
|
|
A control word ends at the first non-letter, so \\pard is not \\par."""
|
|
|
|
def word(name: str) -> int:
|
|
return len(re.findall(rf"\\{name}(?![a-zA-Z])", text))
|
|
|
|
return {
|
|
"paragraph": word("par"),
|
|
"table_row": word("row"),
|
|
"cell": word("cell"),
|
|
"image": word("pict"),
|
|
}
|
|
|
|
|
|
# --- pdf ---------------------------------------------------------------------
|
|
|
|
|
|
def pdf_objects(path: Path) -> dict[str, int] | None:
|
|
"""page and image placements as pdfplumber sees them; None without it."""
|
|
try:
|
|
import pdfplumber
|
|
except ImportError:
|
|
return None
|
|
with pdfplumber.open(str(path)) as pdf:
|
|
pages = len(pdf.pages)
|
|
images = 0
|
|
for page in pdf.pages:
|
|
images += len(page.images)
|
|
page.close()
|
|
return {"page": pages, "image": images}
|
|
|
|
|
|
def pdf_poppler(path: Path) -> dict[str, int] | None:
|
|
"""page (pdfinfo) and image (pdfimages -list, rows of type `image`); None
|
|
when poppler is not installed."""
|
|
info = shutil.which("pdfinfo")
|
|
lister = shutil.which("pdfimages")
|
|
if info is None or lister is None:
|
|
return None
|
|
meta = subprocess.run([info, str(path)], capture_output=True, text=True, check=True).stdout
|
|
pages_match = re.search(r"^Pages:\s+(\d+)", meta, re.MULTILINE)
|
|
listing = subprocess.run(
|
|
[lister, "-list", str(path)], capture_output=True, text=True, check=True
|
|
).stdout
|
|
images = 0
|
|
for line in listing.splitlines()[2:]:
|
|
cells = line.split()
|
|
if len(cells) > 2 and cells[2] == "image":
|
|
images += 1
|
|
return {"page": int(pages_match.group(1)) if pages_match else 0, "image": images}
|
|
|
|
|
|
# --- one file ----------------------------------------------------------------
|
|
|
|
WITNESSED_SUFFIXES = (
|
|
".csv",
|
|
".docx",
|
|
".htm",
|
|
".html",
|
|
".json",
|
|
".md",
|
|
".odt",
|
|
".pdf",
|
|
".pptx",
|
|
".rtf",
|
|
".txt",
|
|
".xlsx",
|
|
".xml",
|
|
)
|
|
|
|
|
|
def witness_file(inbox: Path, path: Path) -> Inventory:
|
|
"""Count one file under `inbox`. Raises WitnessRefused for a file the
|
|
witness does not read."""
|
|
suffix = path.suffix.lower()
|
|
relative = path.relative_to(inbox).as_posix()
|
|
data = path.read_bytes()
|
|
refs: list[str] = []
|
|
sts = False
|
|
if suffix == ".md":
|
|
elements, refs = count_markdown(data.decode("utf-8-sig"))
|
|
witness = "markdown lines"
|
|
elif suffix == ".txt":
|
|
elements, witness = count_text(data.decode("utf-8-sig")), "text lines"
|
|
elif suffix == ".csv":
|
|
elements, witness = count_csv(data.decode("utf-8-sig")), "csv"
|
|
elif suffix == ".json":
|
|
elements, witness = count_json(data.decode("utf-8-sig")), "json"
|
|
elif suffix in (".html", ".htm"):
|
|
parser = _HtmlCounter()
|
|
parser.feed(data.decode("utf-8-sig"))
|
|
parser.close()
|
|
elements, refs, witness = parser.elements, parser.refs, "html.parser"
|
|
elif suffix == ".xml":
|
|
elements, refs, sts = count_sts_xml(data)
|
|
witness = "xml.etree"
|
|
elif suffix == ".docx":
|
|
elements, witness = count_docx(data), "docx zip xml"
|
|
elif suffix == ".pptx":
|
|
elements, witness = count_pptx(data), "pptx zip xml"
|
|
elif suffix == ".xlsx":
|
|
elements, witness = count_xlsx(data), "xlsx zip xml"
|
|
elif suffix == ".odt":
|
|
elements, witness = count_odt(data), "odt zip xml"
|
|
elif suffix == ".rtf":
|
|
elements, witness = count_rtf(data.decode("latin-1")), "rtf control words"
|
|
elif suffix == ".pdf":
|
|
objects = pdf_objects(path)
|
|
if objects is None:
|
|
raise WitnessRefused("pdfplumber is not installed")
|
|
elements, witness = objects, "pdfplumber objects"
|
|
else:
|
|
raise WitnessRefused(f"no witness for {suffix or 'a file without a suffix'}")
|
|
inventory = Inventory(relative, suffix, witness, dict(elements))
|
|
for ref in refs:
|
|
if not ref or _is_remote(ref):
|
|
inventory.images.append(ImageRef(REMOTE, ref))
|
|
else:
|
|
target = resolve_local(inbox, path, ref, sts=sts)
|
|
inventory.images.append(ImageRef(LOCAL, ref, target))
|
|
embedded = elements.get("image", 0) - len(refs)
|
|
inventory.images.extend(ImageRef(EMBEDDED, "") for _ in range(max(0, embedded)))
|
|
return inventory
|
|
|
|
|
|
def walk(inbox: Path) -> Iterator[Path]:
|
|
"""Every file under `inbox`, sorted by relative path, dot-entries skipped."""
|
|
for path in sorted(inbox.rglob("*"), key=lambda p: p.relative_to(inbox).as_posix()):
|
|
if path.is_file() and not any(
|
|
part.startswith(".") for part in path.relative_to(inbox).parts
|
|
):
|
|
yield path
|
|
|
|
|
|
def witness_inbox(inbox: Path) -> dict[str, object]:
|
|
"""The committed fasit form: every document's inventory, every other file
|
|
with the documents that point at it."""
|
|
documents: dict[str, object] = {}
|
|
others: list[str] = []
|
|
pointed: dict[str, list[str]] = {}
|
|
for path in walk(inbox):
|
|
if path.suffix.lower() not in WITNESSED_SUFFIXES:
|
|
others.append(path.relative_to(inbox).as_posix())
|
|
continue
|
|
inventory = witness_file(inbox, path)
|
|
documents[inventory.source_file] = inventory.to_json()
|
|
for ref in inventory.images:
|
|
if ref.target is not None:
|
|
pointed.setdefault(ref.target, [])
|
|
if inventory.source_file not in pointed[ref.target]:
|
|
pointed[ref.target].append(inventory.source_file)
|
|
return {
|
|
"witness_version": WITNESS_VERSION,
|
|
"documents": documents,
|
|
"files": {name: {"pointed_at_by": pointed.get(name, [])} for name in others},
|
|
}
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
import argparse
|
|
|
|
parser = argparse.ArgumentParser(description=__doc__.split("\n", 1)[0])
|
|
parser.add_argument("inbox", type=Path)
|
|
args = parser.parse_args(argv)
|
|
print(json.dumps(witness_inbox(args.inbox), indent=2, ensure_ascii=False, sort_keys=True))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|