test(accounting): the gate opens the bundle itself (BLOCKER B-1)

An independent review of `0b00de4` found the judge was a calculator over a
report the judged writes: `account()` compared BOOKED NUMBERS with the
witness's counts and never opened a concept file. Reproduced here first --
a report that changes not one byte of the bundle and books every element as
carried gave `GATE GREEN`, exit 0, and so did booking every element as
rejected.

The witness now gives every element THE PIECES OF TEXT IT IS MADE OF, and
the gate looks for each of them in the concept bodies the run wrote. Pieces
and not one joined string: a reader writes a heading's marker and a
picture's pointer block between the parts of a container, so a section is
never one contiguous run even when every word of it is there.

Also in the judge, each with a test driving it from both sides:

- a negative booking, a document declared persisted that no concept names,
  a document declared rejected that the bundle holds, a rejection code
  outside a closed list, and an `accounting_version` the gate does not read
  are each REFUSED rather than summed;
- a document the build PERSISTED whose report carries nothing from it is
  never clean ("everything rejected" was);
- an asset proves a carry only when its BYTES hash to the source's and it
  stands under the name the layout gives it. The check was a name check, so
  a zero-byte file called `<sha12>-x.png` read as a carry (m-1).

NOT ONE ELEMENT COUNT MOVED: the 13 fixture documents' counts are identical
before and after, so this commit changes what the gate CHECKS and nothing
about what the witness counts. `texts` is additive in the committed fasit.

The rtf text scanner reads `\uN` escapes and skips `{\fonttbl}`-class
groups, or a fixture's font table reads as the first paragraph of its prose;
xlsx cell text is resolved through `sharedStrings.xml`, where a
spreadsheet's words actually live; a PDF page carries its own text lines,
which no row could see before.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Kjell Tore Guttormsen 2026-09-18 01:40:48 +02:00
commit 656cbe5d02
Signed by: ktg
SSH key fingerprint: SHA256:JakMjO6FTBBzN0Bhfj9saOoEjaFxlSdYuZQQpM/lF9Q
5 changed files with 2177 additions and 254 deletions

File diff suppressed because it is too large Load diff

View file

@ -17,6 +17,24 @@
}
],
"suffix": ".html",
"texts": {
"cell": [],
"heading": [
[
"Skjult"
]
],
"image": [
[]
],
"list_item": [],
"paragraph": [
[
"Denne teksten bærer et usynlig tegn."
]
],
"table": []
},
"witness": "html.parser"
}
},

View file

@ -22,6 +22,7 @@ from __future__ import annotations
import json
import subprocess
import sys
import tempfile
from pathlib import Path
from typing import Any
@ -135,9 +136,10 @@ def test_the_witness_matches_a_hand_count(name: str) -> None:
def test_a_fenced_heading_is_not_a_heading_to_the_witness() -> None:
elements, _ = witness.count_markdown("# Real\n\n```bash\n# not one\n```\n")
assert elements["heading"] == 1
assert elements["code_block"] == 1
count, _ = witness.count_markdown("# Real\n\n```bash\n# not one\n```\n")
assert count.counts["heading"] == 1
assert count.counts["code_block"] == 1
assert count.texts["heading"] == [["# Real"]]
def test_the_sts_image_reference_resolves_through_the_graphics_directory() -> None:
@ -163,29 +165,52 @@ def test_the_witness_refuses_a_doctype() -> None:
# --- 2. every row can go both ways -------------------------------------------
#: The two headings of `a.md` as a bundle would carry them. The gate verifies
#: a booked `carried` against THIS, never against the declaration.
_BUNDLE_TEXT = (
"# Foerste overskrift\n\nProsa.\n\n"
"![](/assets/0123456789ab-x.png)\n_Source: x.png_\n\n# Andre overskrift\n"
)
def _inventory() -> dict[str, Any]:
return {
"documents": {
"a.md": {"suffix": ".md", "elements": {"heading": 2, "image": 1}, "images": []},
"a.md": {
"suffix": ".md",
"elements": {"heading": 2, "image": 1},
"texts": {
"heading": ["Foerste overskrift", "Andre overskrift"],
"image": [""],
},
"images": [{"kind": "local", "ref": "graphics/x.png", "target": "graphics/x.png"}],
},
},
"files": {"graphics/x.png": {"pointed_at_by": ["a.md"]}},
}
def _assets(*paths: Path) -> dict[str, str]:
"""The `assets/` directory a build wrote for these source files."""
return {f"{gate._sha12(p)}-{p.name}": gate._sha256(p) for p in paths}
def _build(
*,
accounting: dict[str, Any] | None = None,
sources: set[str] | None = None,
assets: set[str] | None = None,
assets: dict[str, str] | None = None,
log: str = "",
exit_code: int = 0,
bundle_text: str = _BUNDLE_TEXT,
) -> gate.Build:
return gate.Build(
exit_code=exit_code,
log=log,
accounting=accounting,
source_files={"a.md"} if sources is None else sources,
asset_prefixes=assets or set(),
assets=assets or {},
bundle_text=bundle_text,
)
@ -280,14 +305,14 @@ def test_row3_is_red_when_no_fate_is_declared(tmp_path: Path) -> None:
def test_a_file_carried_through_a_document_and_rejected_is_double_booked(tmp_path: Path) -> None:
corpus = _corpus(tmp_path)
carried = {gate._sha12(corpus / "graphics" / "x.png")}
carried = _assets(corpus / "graphics" / "x.png")
units = gate.account(_inventory(), _build(accounting=_declared(), assets=carried), corpus)
assert (units[1].unaccounted, units[1].double) == (0, 1)
def test_a_file_carried_through_a_document_and_declared_carried_is_clean(tmp_path: Path) -> None:
corpus = _corpus(tmp_path)
carried = {gate._sha12(corpus / "graphics" / "x.png")}
carried = _assets(corpus / "graphics" / "x.png")
build = _build(accounting=_declared(fate="carried"), assets=carried)
assert gate.account(_inventory(), build, corpus)[1].clean
@ -305,7 +330,7 @@ def test_an_unpointed_file_sharing_bytes_with_a_carried_one_is_not_carried(
(corpus / "graphics" / "twin.png").write_bytes(b"png bytes")
inventory = _inventory()
inventory["files"]["graphics/twin.png"] = {"pointed_at_by": []}
carried = {gate._sha12(corpus / "graphics" / "x.png")}
carried = _assets(corpus / "graphics" / "x.png")
units = gate.account(inventory, _build(assets=carried), corpus)
assert [(u.name, u.double) for u in units[1:]] == [
("graphics/twin.png", 0),
@ -315,11 +340,153 @@ def test_an_unpointed_file_sharing_bytes_with_a_carried_one_is_not_carried(
def test_without_the_door_double_booking_is_derived_from_conservation(tmp_path: Path) -> None:
corpus = _corpus(tmp_path)
carried = {gate._sha12(corpus / "graphics" / "x.png")}
carried = _assets(corpus / "graphics" / "x.png")
assert gate.account(_inventory(), _build(assets=carried), corpus)[1].double == 1
assert gate.account(_inventory(), _build(), corpus)[1].clean
# --- B-1: the judge opens the bundle itself ----------------------------------
#
# Written RED 2026-09-18 against the hardening order. At 864570b the gate
# compared BOOKED NUMBERS with the witness's counts and never opened a concept
# file, so a report that booked every element of every document as carried was
# `GATE GREEN` over a bundle holding nothing (independent review, B-1).
def _all_carried(headings: int = 2, images: int = 1) -> dict[str, Any]:
"""A report that books everything as carried, the cheat's shape."""
return {
"accounting_version": 1,
"documents": [
{
"source_file": "a.md",
"status": "persisted",
"code": None,
"inventory": {"heading": 2, "image": 1},
"fates": {
"heading": {"carried": headings},
"image": {"carried": images},
},
}
],
"files": [
{"source_file": "graphics/x.png", "fate": "rejected", "code": "extractor_unknown"}
],
}
def test_carried_text_the_bundle_does_not_hold_is_unverified(tmp_path: Path) -> None:
build = _build(accounting=_all_carried(images=0), bundle_text="")
unit = gate.account(_inventory(), build, _corpus(tmp_path))[0]
assert unit.unverified == 2
assert not unit.clean
def test_carried_text_the_bundle_holds_verifies(tmp_path: Path) -> None:
corpus = _corpus(tmp_path)
build = _build(accounting=_all_carried(images=0), assets=_assets(corpus / "graphics" / "x.png"))
unit = gate.account(_inventory(), build, corpus)[0]
assert (unit.unverified, unit.verified) == (0, 2)
def test_one_heading_carried_of_two_in_the_bundle_is_unverified(tmp_path: Path) -> None:
build = _build(accounting=_all_carried(images=0), bundle_text="# Foerste overskrift\n")
unit = gate.account(_inventory(), build, _corpus(tmp_path))[0]
assert unit.unverified == 1
def test_an_image_booked_carried_without_its_bytes_is_unverified(tmp_path: Path) -> None:
"""The image element has no text of its own, so the only proof it was
carried is the asset. Without it the booking is not verifiable, and an
unverifiable booking is never clean."""
unit = gate.account(_inventory(), _build(accounting=_all_carried()), _corpus(tmp_path))[0]
assert unit.unverified >= 1
assert not unit.clean
def test_a_negative_booking_is_never_clean(tmp_path: Path) -> None:
declared = _all_carried()
declared["documents"][0]["fates"]["heading"] = {"carried": 25, "rejected": {"x": -15}}
unit = gate.account(_inventory(), _build(accounting=declared), _corpus(tmp_path))[0]
assert unit.invalid >= 1
assert not unit.clean
def test_a_document_declared_persisted_that_is_not_in_the_bundle_is_never_clean(
tmp_path: Path,
) -> None:
build = _build(accounting=_all_carried(), sources=set())
unit = gate.account(_inventory(), build, _corpus(tmp_path))[0]
assert unit.invalid >= 1
def test_everything_rejected_is_never_clean_for_a_document_the_build_persisted(
tmp_path: Path,
) -> None:
declared = _all_carried()
declared["documents"][0]["fates"] = {
"heading": {"rejected": {"fail_secure": 2}},
"image": {"rejected": {"fail_secure": 1}},
}
unit = gate.account(_inventory(), _build(accounting=declared), _corpus(tmp_path))[0]
assert unit.invalid >= 1
assert "persisted" in "; ".join(unit.notes)
def test_everything_rejected_is_clean_for_a_document_the_build_refused(tmp_path: Path) -> None:
declared = _all_carried()
declared["documents"][0]["status"] = "rejected"
declared["documents"][0]["code"] = "fail_secure"
declared["documents"][0]["fates"] = {
"heading": {"rejected": {"fail_secure": 2}},
"image": {"rejected": {"fail_secure": 1}},
}
unit = gate.account(
_inventory(), _build(accounting=declared, sources=set()), _corpus(tmp_path)
)[0]
assert unit.clean
def test_a_rejection_code_outside_the_closed_list_is_never_clean(tmp_path: Path) -> None:
declared = _all_carried()
declared["documents"][0]["status"] = "rejected"
declared["documents"][0]["code"] = "because_i_said_so"
declared["documents"][0]["fates"] = {
"heading": {"rejected": {"because_i_said_so": 2}},
"image": {"rejected": {"because_i_said_so": 1}},
}
unit = gate.account(
_inventory(), _build(accounting=declared, sources=set()), _corpus(tmp_path)
)[0]
assert unit.invalid >= 1
def test_an_accounting_version_the_gate_does_not_read_is_never_clean(tmp_path: Path) -> None:
declared = _all_carried()
declared["accounting_version"] = 2
units = gate.account(_inventory(), _build(accounting=declared), _corpus(tmp_path))
assert not any(u.clean for u in units)
def test_an_asset_with_the_right_name_and_the_wrong_bytes_is_not_carried(tmp_path: Path) -> None:
"""m-1: the check was a NAME check, so a zero-byte file called
`<sha12>-x.png` proved a carry."""
corpus = _corpus(tmp_path)
source = corpus / "graphics" / "x.png"
lying = {f"{gate._sha12(source)}-x.png": gate._sha256_bytes(b"")}
build = _build(accounting=_declared(fate="carried"), assets=lying)
assert gate.account(_inventory(), build, corpus)[1].unaccounted == 1
def test_the_cheat_that_books_everything_carried_makes_row3_red(tmp_path: Path) -> None:
"""The review's `MODE=carried`: a report that changes not one byte of the
bundle and books every element as carried."""
units = gate.account(
_inventory(), _build(accounting=_all_carried(), bundle_text=""), _corpus(tmp_path)
)
assert gate.row3(units, door=True).status == gate.RED
_HONEST_LOG = (
"* **Images**: 0 carried of 1 found, written to `assets/`.\n"
"* a.md: 3 elements found in the source, 0 carried: document rejected `fail_secure`\n"
@ -417,6 +584,46 @@ def real_rows() -> list[gate.Row]:
return gate.evaluate(r761=None, ci=True, consume=False)
def _cheating_report(inventory: dict[str, Any], mode: str) -> dict[str, Any]:
"""The review's `cheat.py`, as data: a report that changes not one byte of
the bundle and books every element as carried (or as rejected)."""
documents = []
for name, entry in inventory["documents"].items():
if mode == "carried":
fates = {kind: {"carried": n} for kind, n in entry["elements"].items()}
else:
fates = {
kind: {"rejected": {"extractor_unknown": n}} if n else {"rejected": {}}
for kind, n in entry["elements"].items()
}
documents.append(
{
"source_file": name,
"status": "persisted",
"code": None,
"inventory": dict(entry["elements"]),
"fates": fates,
}
)
files = [{"source_file": name, "fate": "carried", "code": None} for name in inventory["files"]]
return {"accounting_version": 1, "documents": documents, "files": files}
@pytest.mark.parametrize("mode", ["carried", "empty"])
def test_a_report_the_build_did_not_write_cannot_make_row3_green(mode: str) -> None:
"""B-1, end to end on the real fixture bundle. Until 2026-09-18 both modes
gave `GATE GREEN`, exit 0: the gate compared the report's numbers with the
witness's and never opened a concept file."""
pytest.importorskip("pdfplumber")
pytest.importorskip("pypandoc")
inventory = gate.load_inventory(gate.INVENTORY)
with tempfile.TemporaryDirectory() as tmp:
build = gate.run_build(gate.CORPUS, Path(tmp), door=True)
build.accounting = _cheating_report(inventory, mode)
units = gate.account(inventory, build, gate.CORPUS)
assert gate.row3(units, door=True).status == gate.RED
def test_the_door_exists() -> None:
assert gate.door_available()
@ -433,4 +640,6 @@ def test_the_real_gate_is_green_on_every_fixture_row(real_rows: list[gate.Row])
(5, gate.GREEN),
(6, gate.SKIPPED),
]
assert real_rows[2].reason.startswith("u = 0 unaccounted, d = 0 double-booked")
assert real_rows[2].reason.startswith(
"u = 0 unaccounted, d = 0 double-booked, 0 booked carried and not in the bundle"
)

View file

@ -64,7 +64,7 @@ import warnings
import zipfile
from collections.abc import Iterable, Mapping
from dataclasses import dataclass, field
from pathlib import Path
from pathlib import Path, PurePosixPath
from typing import Any
TOOLS = Path(__file__).resolve().parent
@ -190,13 +190,20 @@ def door_available() -> bool:
@dataclass
class Build:
"""What one `okf build` run left behind, read back from the artifacts."""
"""What one `okf build` run left behind, read back from the artifacts.
`bundle_text` is every concept BODY the run wrote, joined. It is what
makes this gate a judge rather than a calculator: a booking that says an
element was carried is checked against these bytes, not against the
number beside it.
"""
exit_code: int
log: str
accounting: dict[str, Any] | None
source_files: set[str]
asset_prefixes: set[str]
assets: dict[str, str]
bundle_text: str = ""
def _frontmatter_source_file(text: str) -> str | None:
@ -212,16 +219,31 @@ def _frontmatter_source_file(text: str) -> str | None:
return value
def _body(text: str) -> str:
"""A concept file without its frontmatter block."""
if not text.startswith("---\n"):
return text
end = text.find("\n---\n", 4)
return text[end + 5 :] if end >= 0 else text
def read_bundle(bundle: Path, exit_code: int, accounting_path: Path | None) -> Build:
sources: set[str] = set()
for path in bundle.rglob("*.md"):
bodies: list[str] = []
for path in sorted(bundle.rglob("*.md"), key=lambda p: p.as_posix()):
if "assets" in path.relative_to(bundle).parts:
continue
found = _frontmatter_source_file(path.read_text(encoding="utf-8"))
text = path.read_text(encoding="utf-8")
found = _frontmatter_source_file(text)
if found:
sources.add(found)
assets = bundle / "assets"
prefixes = {p.name[:12] for p in assets.iterdir()} if assets.is_dir() else set()
bodies.append(_body(text))
assets_dir = bundle / "assets"
assets = (
{p.name: _sha256(p) for p in sorted(assets_dir.iterdir()) if p.is_file()}
if assets_dir.is_dir()
else {}
)
log_path = bundle / "log.md"
accounting = None
if accounting_path is not None and accounting_path.is_file():
@ -231,7 +253,8 @@ def read_bundle(bundle: Path, exit_code: int, accounting_path: Path | None) -> B
log=log_path.read_text(encoding="utf-8") if log_path.is_file() else "",
accounting=accounting,
source_files=sources,
asset_prefixes=prefixes,
assets=assets,
bundle_text="\n".join(bodies),
)
@ -269,8 +292,111 @@ def run_build(corpus: Path, workdir: Path, *, door: bool, gate: str | None = Non
return read_bundle(bundle, code, accounting_path)
def _sha256_bytes(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def _sha256(path: Path) -> str:
return _sha256_bytes(path.read_bytes())
def _sha12(path: Path) -> str:
return hashlib.sha256(path.read_bytes()).hexdigest()[:12]
return _sha256(path)[:12]
def asset_holds(build: Build, source: Path) -> bool:
"""Did the run carry THESE bytes, under the name the layout gives them?
Both halves are load-bearing. The name alone was the check until an
independent review wrote a zero-byte file called `<sha12>-x.png` and the
gate read it as a carry (m-1); the bytes alone would credit a file the
layout would have named something else.
"""
digest = _sha256(source)
return build.assets.get(f"{digest[:12]}-{source.name}") == digest
# --- the judge's own reading of the bundle ------------------------------------
#: Every rejection code this build can emit, read off the package's own source
#: 2026-09-18, plus the guard's two dispositions. A code outside this list is
#: RED: a report gets to say WHY it dropped something, not to invent the
#: vocabulary it says it in.
REJECTION_CODES: frozenset[str] = frozenset(
{
"asset_collision",
"asset_pdf_unsupported",
"asset_remote",
"asset_samples_invalid",
"asset_too_large",
"asset_type_unknown",
"asset_unresolved",
"extractor_binary_missing",
"extractor_binary_version",
"extractor_convert_error",
"extractor_decode_error",
"extractor_empty_conversion",
"extractor_empty_csv",
"extractor_empty_pdf",
"extractor_extra_missing",
"extractor_id",
"extractor_ocr_group_missing",
"extractor_pdf_error",
"extractor_unknown",
"extractor_version",
"extractor_xml_doctype",
"extractor_xml_parse_error",
"fail_secure",
"inbox_gate",
"inbox_slug_collision",
"inbox_slug_empty",
"inbox_slug_too_long",
"inbox_source_file_invalid",
"inbox_source_file_unaddressable",
"inbox_source_title_unaddressable",
"inbox_title_invalid",
"inventory_error",
"inventory_unreadable",
"quarantine_review",
}
)
_NOT_ALNUM = re.compile(r"[\W_]+")
#: A converter attribute block (`{.mark}`, `{#slide-1}`). Its letters stand
#: between words that WERE carried, so it is removed from the bundle text
#: before comparing -- the same allowance the build makes, and no looser.
_CONVERTER_ATTRIBUTE = re.compile(r"\{[#.][^{}\n]*\}")
def _norm(text: str) -> str:
return _NOT_ALNUM.sub("", text.casefold())
class Finder:
"""Is this piece of the SOURCE in the text the bundle holds?
Searched forward from the last hit first, because a reader keeps the
source's order; a miss there falls back to the whole text, so an element
that moved is still found. The gate owns this code: borrowing the build's
own finder would make the judge agree with the judged by construction.
"""
def __init__(self, text: str) -> None:
self.text = _norm(_CONVERTER_ATTRIBUTE.sub("", text))
self.cursor = 0
def __call__(self, piece: str) -> bool:
needle = _norm(piece)
if not needle:
return False
at = self.text.find(needle, self.cursor)
if at < 0:
at = self.text.find(needle)
if at < 0:
return False
self.cursor = at + len(needle)
return True
# --- accounting --------------------------------------------------------------
@ -278,79 +404,261 @@ def _sha12(path: Path) -> str:
@dataclass
class Unit:
"""One inventoried thing: a document, or an inbox file that is not one."""
"""One inventoried thing: a document, or an inbox file that is not one.
`unaccounted` and `double` are about the NUMBERS; `unverified` and
`invalid` are about the bundle and the declaration themselves. A booking
the gate could not verify is never clean, and `verified`/`unverifiable`
carry the denominator behind that word.
"""
name: str
kind: str
unaccounted: int
double: int
unverified: int = 0
invalid: int = 0
verified: int = 0
unverifiable: int = 0
notes: list[str] = field(default_factory=list)
@property
def clean(self) -> bool:
return self.unaccounted == 0 and self.double == 0
return not (self.unaccounted or self.double or self.unverified or self.invalid)
def _conservation_held(build: Build) -> bool:
return build.exit_code == 0 and "K1b FAILED" not in build.log
def _image_proof(
images: Iterable[Mapping[str, Any]], build: Build, corpus: Path, finder: Finder
) -> tuple[int, int, int]:
"""(bytes proved in `assets/`, references found as text, images the gate
cannot check).
An image element carries no text of its own, so the only proof it was
carried is the asset -- and the only proof a POINTER survived is the
reference standing in the bundle. An image embedded in a binary container
has no source file to hash, so the gate says it cannot check it rather
than passing it.
"""
proved = pointed = blind = 0
for image in images:
kind = image.get("kind")
target = image.get("target")
ref = str(image.get("ref") or "")
names = [n for n in (ref, PurePosixPath(ref).name if ref else "") if n]
if target:
names.append(PurePosixPath(target).name)
if asset_holds(build, corpus / target):
proved += 1
elif kind == witness.EMBEDDED:
blind += 1
if any(finder(name) for name in names):
pointed += 1
return proved, pointed, blind
def _document_unit(
name: str,
entry: Mapping[str, Any],
declared: Mapping[str, Any] | None,
build: Build,
corpus: Path,
version_ok: bool,
) -> Unit:
"""One document's account, checked against the bundle the run wrote."""
elements: dict[str, int] = dict(entry["elements"])
texts: Mapping[str, list[str]] = entry.get("texts", {})
images: list[Mapping[str, Any]] = list(entry.get("images", []))
total = sum(elements.values())
if declared is None:
return Unit(name, "document", total, 0, notes=["no declared fates"])
if not version_ok:
return Unit(
name,
"document",
total,
0,
invalid=1,
notes=[
f"accounting_version is not {ACCOUNTING_VERSION}: the declaration is unreadable"
],
)
fates: Mapping[str, Any] = declared.get("fates", {})
persisted = name in build.source_files
unaccounted = double = unverified = invalid = verified = unverifiable = 0
notes: list[str] = []
status = declared.get("status")
code = declared.get("code")
if status == "persisted" and not persisted:
invalid += 1
notes.append("declared persisted; no concept in the bundle names this document")
if status == "rejected" and persisted:
invalid += 1
notes.append("declared rejected; the bundle holds a concept from this document")
if code is not None and code not in REJECTION_CODES:
invalid += 1
notes.append(f"document code `{code}` is not one this gate knows")
finder = Finder(build.bundle_text)
booked_carried = 0
for element in sorted(set(elements) | set(fates)):
fate: Mapping[str, Any] = fates.get(element, {})
carried = int(fate.get("carried", 0))
pointer = int(fate.get("pointer", 0))
rejected = {str(k): int(v) for k, v in (fate.get("rejected") or {}).items()}
have = elements.get(element, 0)
booked_carried += max(carried, 0) + max(pointer, 0)
if min([carried, pointer, *rejected.values()], default=0) < 0:
invalid += 1
notes.append(f"{element}: a negative booking {json.dumps(fate, sort_keys=True)}")
unknown = sorted(c for c in rejected if c not in REJECTION_CODES)
if unknown:
invalid += 1
notes.append(
f"{element}: rejection code(s) {', '.join(unknown)} outside the closed list"
)
booked = carried + pointer + sum(rejected.values())
if booked < have:
unaccounted += have - booked
notes.append(f"{element}: {booked} booked of {have}")
elif booked > have:
double += booked - have
notes.append(f"{element}: {booked} booked, source has {have}")
want = max(carried, 0) + max(pointer, 0)
if want == 0:
continue
if not persisted:
invalid += 1
notes.append(
f"{element}: {want} booked carried, but the bundle holds nothing from here"
)
if element == "image":
proved, pointed, blind = _image_proof(images, build, corpus, finder)
reach = min(carried, proved) + min(pointer, pointed)
verified += reach
short = want - reach
if short > 0:
take = min(short, blind)
unverifiable += take
if short > take:
unverified += short - take
notes.append(
f"image: {want} booked, {proved} asset(s) with the source's bytes and "
f"{pointed} reference(s) found in the bundle"
)
continue
# An element is made of PIECES, and it is carried only if EVERY one of
# them is in the bundle: a reader writes its own markers between the
# parts of a container, so a section is never one contiguous run.
elements_pieces = [
[piece for piece in element_pieces if _norm(piece)]
for element_pieces in texts.get(element, [])
]
sayable = [pieces for pieces in elements_pieces if pieces]
found = sum(1 for pieces in sayable if all(finder(piece) for piece in pieces))
verified += min(found, want)
if want > found:
short = want - found
take = min(short, len(elements_pieces) - len(sayable))
unverifiable += take
if short > take:
unverified += short - take
notes.append(
f"{element}: {want} booked carried, {found} of {len(sayable)} "
"found in the bundle"
)
if persisted and total > 0 and booked_carried == 0:
invalid += 1
notes.append("the build persisted this document and the report carries nothing from it")
return Unit(
name,
"document",
unaccounted,
double,
unverified=unverified,
invalid=invalid,
verified=verified,
unverifiable=unverifiable,
notes=notes,
)
def _file_unit(
name: str,
entry: Mapping[str, Any],
declared: Mapping[str, Any] | None,
build: Build,
corpus: Path,
version_ok: bool,
) -> Unit:
"""One inbox file that is not a document the build reads."""
pointed_by = entry["pointed_at_by"]
# Bytes in assets/ prove a carry only for a file a document points at:
# an unpointed file with the same bytes (R761 ships 8 such duplicates)
# was not carried through anything.
carried = bool(pointed_by) and asset_holds(build, corpus / name)
merged = name in build.source_files
notes: list[str] = []
invalid = 0
if declared is not None and not version_ok:
return Unit(
name,
"file",
1,
0,
invalid=1,
notes=[
f"accounting_version is not {ACCOUNTING_VERSION}: the declaration is unreadable"
],
)
false_claim = False
if declared is not None:
fate = declared.get("fate")
code = declared.get("code")
rejected = fate == "rejected"
false_claim = fate == "carried" and not carried
if false_claim:
notes.append("declared carried; no asset holds this file's bytes under its own name")
if code is not None and code not in REJECTION_CODES:
invalid += 1
notes.append(f"file code `{code}` is not one this gate knows")
else:
# K1b: every walked file is merged or a coded rejection, so a file
# that is not merged was booked as a rejection.
rejected = not merged and _conservation_held(build)
fates = sum((carried, merged, rejected))
if carried and rejected:
notes.append(f"carried via {', '.join(pointed_by) or 'a document'} AND rejected")
unaccounted = 1 if fates == 0 or false_claim else 0
return Unit(
name,
"file",
unaccounted,
max(0, fates - 1),
invalid=invalid,
verified=1 if carried else 0,
notes=notes,
)
def account(inventory: Mapping[str, Any], build: Build, corpus: Path) -> list[Unit]:
"""Give every inventoried element and file its fate, or say it has none."""
declared_docs = {}
declared_files = {}
"""Give every inventoried element and file its fate, and CHECK it."""
declared_docs: dict[str, Any] = {}
declared_files: dict[str, Any] = {}
version_ok = True
if build.accounting is not None:
declared_docs = {d["source_file"]: d for d in build.accounting.get("documents", [])}
declared_files = {f["source_file"]: f for f in build.accounting.get("files", [])}
version_ok = build.accounting.get("accounting_version") == ACCOUNTING_VERSION
units: list[Unit] = []
for name, entry in sorted(inventory["documents"].items()):
elements: dict[str, int] = entry["elements"]
declared = declared_docs.get(name)
if declared is None:
units.append(Unit(name, "document", sum(elements.values()), 0, ["no declared fates"]))
continue
unaccounted = double = 0
notes: list[str] = []
fates: dict[str, Any] = declared.get("fates", {})
for element in sorted(set(elements) | set(fates)):
fate = fates.get(element, {})
booked = (
int(fate.get("carried", 0))
+ int(fate.get("pointer", 0))
+ sum(int(v) for v in fate.get("rejected", {}).values())
)
have = elements.get(element, 0)
if booked < have:
unaccounted += have - booked
notes.append(f"{element}: {booked} booked of {have}")
elif booked > have:
double += booked - have
notes.append(f"{element}: {booked} booked, source has {have}")
units.append(Unit(name, "document", unaccounted, double, notes))
units.append(
_document_unit(name, entry, declared_docs.get(name), build, corpus, version_ok)
)
for name, entry in sorted(inventory["files"].items()):
pointed_by = entry["pointed_at_by"]
# Bytes in assets/ prove a carry only for a file a document points at:
# an unpointed file with the same bytes (R761 ships 8 such duplicates)
# was not carried through anything.
carried = bool(pointed_by) and _sha12(corpus / name) in build.asset_prefixes
merged = name in build.source_files
declared = declared_files.get(name)
notes = []
false_claim = False
if declared is not None:
rejected = declared.get("fate") == "rejected"
false_claim = declared.get("fate") == "carried" and not carried
if false_claim:
notes.append("declared carried, bytes absent from assets/")
else:
# K1b: every walked file is merged or a coded rejection, so a file
# that is not merged was booked as a rejection.
rejected = not merged and _conservation_held(build)
fates = sum((carried, merged, rejected))
if carried and rejected:
notes.append(f"carried via {', '.join(pointed_by) or 'a document'} AND rejected")
unaccounted = 1 if fates == 0 or false_claim else 0
units.append(Unit(name, "file", unaccounted, max(0, fates - 1), notes))
units.append(_file_unit(name, entry, declared_files.get(name), build, corpus, version_ok))
return units
@ -405,20 +713,45 @@ def row2(table: Iterable[str], inventory: Mapping[str, Any], build: Build, door:
return _row(2, name, len(good), len(table), reason, details)
def _tally(units: Iterable[Unit]) -> str:
"""What the gate FOUND, with the denominator beside it."""
units = list(units)
verified = sum(u.verified for u in units)
unverified = sum(u.unverified for u in units)
blind = sum(u.unverifiable for u in units)
return (
f"{verified} carried element(s) found in the bundle, {unverified} claimed and not found, "
f"{blind} carrying no text of their own (the gate cannot check those)"
)
def row3(units: list[Unit], door: bool) -> Row:
clean = sum(1 for u in units if u.clean)
u_total = sum(u.unaccounted for u in units)
d_total = sum(u.double for u in units)
reason = f"u = {u_total} unaccounted, d = {d_total} double-booked"
unverified = sum(u.unverified for u in units)
invalid = sum(u.invalid for u in units)
reason = (
f"u = {u_total} unaccounted, d = {d_total} double-booked, "
f"{unverified} booked carried and not in the bundle, {invalid} declaration(s) the gate refuses"
)
if not door:
reason += f"; no `{ACCOUNTING_FLAG}` door, so no element has a declared fate"
details = [
f"{u.kind} {u.name}: u={u.unaccounted} d={u.double}"
details = [_tally(units)] + [
f"{u.kind} {u.name}: u={u.unaccounted} d={u.double} "
f"unverified={u.unverified} invalid={u.invalid}"
+ (f" ({'; '.join(u.notes)})" if u.notes else "")
for u in units
if not u.clean
]
return _row(3, "accounting after build (u = 0 and d = 0)", clean, len(units), reason, details)
return _row(
3,
"accounting after build (u = 0, d = 0, and every carried element found)",
clean,
len(units),
reason,
details,
)
def row4(inventory: Mapping[str, Any], build: Build) -> Row:
@ -480,16 +813,23 @@ def row5(pairs: list[tuple[str, list[str]]], notes: list[str]) -> Row:
return _row(5, "two witnesses agree", good, len(pairs), reason, details)
def _counts(count: witness.Count | None) -> dict[str, int] | None:
return None if count is None else dict(count.counts)
def witness_pairs(r761: Path | None) -> tuple[list[tuple[str, list[str]]], list[str]]:
sts_xml, _, _ = witness.count_sts_xml(STS_FIXTURE.read_bytes())
pairs = [
(
"sts fixture (xml | json)",
compare(sts_xml, witness.count_sts_json(STS_TWIN.read_bytes())),
compare(_counts(sts_xml), _counts(witness.count_sts_json(STS_TWIN.read_bytes()))),
),
(
"pdf fixture (pdfplumber | poppler)",
compare(witness.pdf_objects(PDF_FIXTURE), witness.pdf_poppler(PDF_FIXTURE)),
compare(
_counts(witness.pdf_objects(PDF_FIXTURE)),
_counts(witness.pdf_poppler(PDF_FIXTURE)),
),
),
]
notes: list[str] = []
@ -502,14 +842,17 @@ def witness_pairs(r761: Path | None) -> tuple[list[tuple[str, list[str]]], list[
pairs.append(
(
"R761 sts (xml | json)",
compare(r761_xml, witness.count_sts_json((r761 / R761_JSON).read_bytes())),
compare(
_counts(r761_xml),
_counts(witness.count_sts_json((r761 / R761_JSON).read_bytes())),
),
)
)
pdf = r761 / R761_PDF
pairs.append(
(
"R761 pdf (pdfplumber | poppler)",
compare(witness.pdf_objects(pdf), witness.pdf_poppler(pdf)),
compare(_counts(witness.pdf_objects(pdf)), _counts(witness.pdf_poppler(pdf))),
)
)
return pairs, notes
@ -533,7 +876,7 @@ def row6(r761: Path | None, ci: bool) -> Row:
inbox = Path(tmp) / "inbox"
archive.extractall(inbox)
inventory = witness.witness_inbox(inbox)
json_counts = witness.count_sts_json((r761 / R761_JSON).read_bytes())
json_counts = dict(witness.count_sts_json((r761 / R761_JSON).read_bytes()).counts)
for index, gate_name in enumerate(R761_GATES):
build = run_build(inbox, Path(tmp) / f"work{index}", door=door, gate=gate_name)
runs.append((gate_name or "default", build, account(inventory, build, inbox)))

View file

@ -44,10 +44,11 @@ import re
import shutil
import subprocess
import zipfile
from collections.abc import Iterator
from collections.abc import Iterable, Iterator, Mapping
from dataclasses import dataclass, field
from html.parser import HTMLParser
from pathlib import Path, PurePosixPath
from typing import Any
from xml.etree import ElementTree as ET
WITNESS_VERSION = 1
@ -76,6 +77,40 @@ class ImageRef:
target: str | None = None
class Count:
"""Elements of one file: how many of each, and THE TEXT OF EACH.
The text is what makes the gate a judge. A count alone can only be
compared with another count, so a report claiming an element was carried
could never be checked against the bundle; with the element's own text
the gate looks for it and says whether it is there.
An element that carries no text of its own (a picture, a spreadsheet's
sheet) gets the empty string, and the gate reports it as one it cannot
check rather than as one that passed.
"""
def __init__(self, vocabulary: Iterable[str]) -> None:
self.vocabulary = tuple(vocabulary)
self.counts: dict[str, int] = dict.fromkeys(self.vocabulary, 0)
self.texts: dict[str, list[list[str]]] = {name: [] for name in self.vocabulary}
def add(self, role: str, *pieces: str) -> list[str]:
"""Count one element and keep the PIECES of text it is made of.
Pieces, not one joined string: a reader writes a heading's marker and
a picture's pointer block between the parts of a container, so a
section's text is not a contiguous run in the bundle even when every
word of it is there. Each piece is looked for on its own.
"""
if role not in self.counts:
raise KeyError(f"{role!r} is not in this format's vocabulary")
kept = [piece for piece in pieces if piece and piece.strip()]
self.counts[role] += 1
self.texts[role].append(kept)
return kept
@dataclass
class Inventory:
"""What one source file holds, element type by element type."""
@ -85,6 +120,7 @@ class Inventory:
witness: str
elements: dict[str, int] = field(default_factory=dict)
images: list[ImageRef] = field(default_factory=list)
texts: dict[str, list[list[str]]] = field(default_factory=dict)
@property
def total(self) -> int:
@ -95,6 +131,10 @@ class Inventory:
"suffix": self.suffix,
"witness": self.witness,
"elements": dict(sorted(self.elements.items())),
"texts": {
name: [list(pieces) for pieces in values]
for name, values in sorted(self.texts.items())
},
"images": [
{"kind": ref.kind, "ref": ref.ref, "target": ref.target} for ref in self.images
],
@ -132,33 +172,36 @@ def resolve_local(inbox: Path, document: Path, ref: str, *, sts: bool = False) -
# --- markdown / text ---------------------------------------------------------
_FENCE_OPEN = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$")
_ATX = re.compile(r"^ {0,3}#{1,6}(\s|$)")
_ATX_LINE = re.compile(r"^ {0,3}#{1,6}(\s|$)")
_DELIMITER_ROW = re.compile(r"^\s*\|?\s*:?-{3,}:?\s*(\|\s*:?-{3,}:?\s*)*\|?\s*$")
_MD_IMAGE = re.compile(r"!\[[^\]]*\]\(\s*<?([^)\s>]+)>?[^)]*\)")
def _unfenced(lines: list[str]) -> tuple[list[str | None], int]:
"""Lines with fenced ones replaced by None, and the number of fences.
def _unfenced(lines: list[str]) -> tuple[list[str | None], list[list[str]]]:
"""Lines with fenced ones replaced by None, and one block per OPENING fence.
CommonMark SS 4.5 in the parts that decide which lines are fenced: up to
three leading spaces, a backtick info string may not hold a backtick, the
closing fence is the same character and at least as long, and an unclosed
fence runs to the end of the text.
fence runs to the end of the text. The blocks are kept apart per opener,
because two fences may stand on consecutive lines and a run of fenced
lines would then read as one.
"""
out: list[str | None] = []
fences = 0
blocks: list[list[str]] = []
opener: str | None = None
for line in lines:
if opener is None:
match = _FENCE_OPEN.match(line)
if match and not (match.group(1)[0] == "`" and "`" in match.group(2)):
opener = match.group(1)
fences += 1
blocks.append([line])
out.append(None)
continue
out.append(line)
continue
out.append(None)
blocks[-1].append(line)
stripped = line.strip()
if (
stripped
@ -167,116 +210,159 @@ def _unfenced(lines: list[str]) -> tuple[list[str | None], int]:
and len(line) - len(line.lstrip(" ")) <= 3
):
opener = None
return out, fences
return out, blocks
def count_markdown(text: str) -> tuple[dict[str, int], list[str]]:
MARKDOWN = ("code_block", "heading", "image", "paragraph", "table", "table_row")
def count_markdown(text: str) -> tuple[Count, list[str]]:
"""heading: ATX lines outside a fence. table: a pipe row followed by a
delimiter row. table_row: the body rows under it. image: `![..](..)`
outside a fence. code_block: a fence. paragraph: a run of non-blank lines
outside a fence that holds none of the above."""
lines, fences = _unfenced(text.split("\n"))
elements = {
"heading": 0,
"paragraph": 0,
"table": 0,
"table_row": 0,
"image": 0,
"code_block": fences,
}
outside a fence that holds none of the above.
Each element's text is the line, or the lines, it is made of."""
lines, blocks = _unfenced(text.split("\n"))
count = Count(MARKDOWN)
refs: list[str] = []
in_table = False
in_paragraph = False
paragraph: list[str] = []
table_header: str | None = None
table_rows: list[str] = []
delimiter_rows: set[int] = set()
def _close_table() -> None:
nonlocal table_header
if table_header is not None:
count.add("table", table_header, *table_rows)
table_header = None
table_rows.clear()
def close_paragraph() -> None:
if paragraph:
count.add("paragraph", *paragraph)
paragraph.clear()
for index, line in enumerate(lines):
if index in delimiter_rows:
continue
if line is None or not line.strip():
in_table = False
in_paragraph = False
if in_table:
in_table = False
_close_table()
close_paragraph()
continue
if in_table:
if "|" in line:
elements["table_row"] += 1
count.add("table_row", line)
table_rows.append(line)
continue
in_table = False
_close_table()
following = lines[index + 1] if index + 1 < len(lines) else None
if "|" in line and following is not None and _DELIMITER_ROW.match(following):
elements["table"] += 1
_close_table()
table_header = line
in_table = True
delimiter_rows.add(index + 1) # the delimiter row is not a body row
in_paragraph = False
close_paragraph()
continue
if _ATX.match(line):
elements["heading"] += 1
in_paragraph = False
if _ATX_LINE.match(line):
count.add("heading", line)
close_paragraph()
continue
found = _MD_IMAGE.findall(line)
if found:
elements["image"] += len(found)
refs.extend(found)
if not _MD_IMAGE.sub("", line).strip():
in_paragraph = False
continue
if not in_paragraph:
elements["paragraph"] += 1
in_paragraph = True
return elements, refs
rest = _MD_IMAGE.sub("", line) if found else line
for ref in found:
count.add("image")
refs.append(ref)
if found and not rest.strip():
close_paragraph()
continue
paragraph.append(rest)
close_paragraph()
_close_table()
for block in blocks:
count.add("code_block", "\n".join(block))
return count, refs
def count_text(text: str) -> dict[str, int]:
TEXT = ("line", "paragraph")
def count_text(text: str) -> Count:
"""paragraph: a run of non-blank lines. line: a non-blank line."""
paragraphs = 0
lines = 0
previous_blank = True
for line in text.split("\n"):
count = Count(TEXT)
block: list[str] = []
for line in [*text.split("\n"), ""]:
if line.strip():
lines += 1
if previous_blank:
paragraphs += 1
previous_blank = False
else:
previous_blank = True
return {"paragraph": paragraphs, "line": lines}
count.add("line", line)
block.append(line)
elif block:
count.add("paragraph", *block)
block = []
return count
def count_csv(text: str) -> dict[str, int]:
CSV = ("cell", "header_cell", "row")
def count_csv(text: str) -> Count:
"""header_cell: cells of the first row. row / cell: every row after it."""
count = Count(CSV)
rows = [row for row in csv.reader(io.StringIO(text)) if row]
if not rows:
return {"header_cell": 0, "row": 0, "cell": 0}
return {
"header_cell": len(rows[0]),
"row": len(rows) - 1,
"cell": sum(len(row) for row in rows[1:]),
}
for position, row in enumerate(rows):
if position == 0:
for value in row:
count.add("header_cell", value)
continue
for value in row:
count.add("cell", value)
count.add("row", *row)
return count
def count_json(text: str) -> dict[str, int]:
"""key: an object member. value: a leaf (string, number, boolean, null)."""
counts = {"key": 0, "value": 0}
JSON = ("key", "value")
def count_json(text: str) -> Count:
"""key: an object member. value: a leaf (string, number, boolean, null).
A key's text is the key; a leaf's text is the leaf as JSON writes it,
which is the form a bundle carrying the document verbatim holds."""
count = Count(JSON)
def walk(node: object) -> None:
if isinstance(node, dict):
counts["key"] += len(node)
for child in node.values():
for key, child in node.items():
count.add("key", key)
walk(child)
elif isinstance(node, list):
for child in node:
walk(child)
else:
counts["value"] += 1
count.add("value", json.dumps(node, ensure_ascii=False))
walk(json.loads(text))
return counts
return count
# --- html --------------------------------------------------------------------
HTML = ("cell", "heading", "image", "list_item", "paragraph", "table")
#: Tags that close themselves: an unclosed `<img>` must not swallow the rest
#: of the document as its own text.
_HTML_VOID = frozenset(
{"img", "br", "hr", "meta", "link", "input", "col", "area", "base", "wbr", "source"}
)
class _HtmlCounter(HTMLParser):
"""heading: h1-h6. paragraph: p. list_item: li. table: table. cell: td,
th. image: img."""
th. image: img. An element's text is the text between its own tags."""
_ROLES = {
**{f"h{level}": "heading" for level in range(1, 7)},
@ -290,16 +376,46 @@ class _HtmlCounter(HTMLParser):
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.elements = {role: 0 for role in sorted(set(self._ROLES.values()))}
self.count = Count(HTML)
self.refs: list[str] = []
self._open: list[tuple[str, str, list[str]]] = []
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
role = self._ROLES.get(tag)
if tag == "img":
if role is not None:
self.count.add(role)
self.refs.append(dict(attrs).get("src") or "")
return
if role is None:
return
self.elements[role] += 1
if tag == "img":
self.refs.append(dict(attrs).get("src") or "")
self._open.append((tag, role, []))
def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
self.handle_starttag(tag, attrs)
def handle_endtag(self, tag: str) -> None:
if tag in _HTML_VOID or self._ROLES.get(tag) is None:
return
for position in range(len(self._open) - 1, -1, -1):
if self._open[position][0] == tag:
_, role, pieces = self._open.pop(position)
kept = self.count.add(role, *pieces)
for outer in self._open:
outer[2].extend(kept)
return
def handle_data(self, data: str) -> None:
if self._open:
self._open[-1][2].append(_normal(data))
def finish(self) -> None:
"""Whatever the document left open still counts, innermost first."""
while self._open:
_, role, pieces = self._open.pop()
kept = self.count.add(role, *pieces)
for outer in self._open:
outer[2].extend(kept)
# --- xml / sts ---------------------------------------------------------------
@ -358,51 +474,73 @@ def _sts_role(tag: str, parent: str | None, grandparent: str | None) -> str | No
return None
def count_sts_xml(data: bytes) -> tuple[dict[str, int], list[str], bool]:
"""Element roles of an STS document; `element` alone for other XML."""
def _normal(text: str) -> str:
return " ".join(text.split())
def count_sts_xml(data: bytes) -> tuple[Count, list[str], bool]:
"""Element roles of an STS document; `element` alone for other XML.
An element's text is its own subtree, whitespace-folded -- the form a
reader writing markdown produces, and the only form in which a container
section can be looked for at all."""
if b"<!DOCTYPE" in data:
raise WitnessRefused("a DOCTYPE is not parsed")
root = ET.fromstring(data)
sts = _local(root.tag) == "standard" or any(_local(el.tag) == "sec" for el in root.iter())
if not sts:
return {"element": sum(1 for _ in root.iter())}, [], False
elements = {role: 0 for role in STS_ROLES}
count = Count(("element",))
for node in root.iter():
pieces = [_normal(t) for t in node.itertext() if t.strip()]
count.add("element", *pieces)
return count, [], False
count = Count(STS_ROLES)
refs: list[str] = []
def walk(node: ET.Element, parent: str | None, grandparent: str | None) -> None:
def walk(node: ET.Element, parent: str | None, grandparent: str | None) -> list[str]:
tag = _local(node.tag)
role = _sts_role(tag, parent, grandparent)
if role is not None:
elements[role] += 1
pieces: list[str] = []
if node.text and node.text.strip():
pieces.append(_normal(node.text))
if role == "image":
href = next((value for key, value in node.attrib.items() if _local(key) == "href"), "")
refs.append(href)
for child in node:
walk(child, tag, parent)
pieces.extend(walk(child, tag, parent))
if child.tail and child.tail.strip():
pieces.append(_normal(child.tail))
if role is not None:
count.add(role, *pieces)
return pieces
walk(root, None, None)
return elements, refs, True
return count, refs, True
def count_sts_json(data: bytes) -> dict[str, int]:
def count_sts_json(data: bytes) -> Count:
"""The same roles, read from the publisher's JSON node tree."""
document = json.loads(data)
elements = {role: 0 for role in STS_ROLES}
count = Count(STS_ROLES)
def walk(node: dict[str, object], parent: str | None, grandparent: str | None) -> None:
def walk(node: Mapping[str, Any], parent: str | None, grandparent: str | None) -> list[str]:
pieces: list[str] = []
if node.get("t") and str(node["t"]).strip():
pieces.append(_normal(str(node["t"])))
body = node.get("x")
if not isinstance(body, dict):
return
return pieces
tag = str(body.get("tag"))
role = _sts_role(tag, parent, grandparent)
if role is not None:
elements[role] += 1
for child in body.get("c") or []:
walk(child, tag, parent)
pieces.extend(walk(child, tag, parent))
if role is not None:
count.add(role, *pieces)
return pieces
for child in document["standardContent"]["c"]:
walk(child, None, None)
return elements
return count
# --- office zips -------------------------------------------------------------
@ -422,159 +560,382 @@ def _text_of(node: ET.Element, tag: str) -> str:
return "".join(t.text or "" for t in node.iter(tag))
def count_docx(data: bytes) -> dict[str, int]:
DOCX = ("cell", "footnote", "heading", "image", "paragraph", "table")
def _docx_lines(para: ET.Element) -> list[str]:
"""A paragraph's text, cut where the document itself breaks a line.
`w:br` and `w:cr` are line boundaries in the format, so the two halves of
a broken paragraph can land in different places -- inside a grid table
they land on different rows, with other cells' text between them."""
lines = [""]
for node in para.iter():
if node.tag == f"{_W}t":
lines[-1] += node.text or ""
elif node.tag in (f"{_W}br", f"{_W}cr"):
lines.append("")
return [_normal(line) for line in lines if line.strip()]
def count_docx(data: bytes) -> Count:
"""heading: a w:p whose style is a heading or title style. paragraph: any
other w:p with text. table: w:tbl. cell: w:tc. image: a:blip.
footnote: a w:footnote with a positive id."""
count = Count(DOCX)
with zipfile.ZipFile(io.BytesIO(data)) as archive:
root = ET.fromstring(archive.read("word/document.xml"))
footnotes = 0
notes = None
if "word/footnotes.xml" in archive.namelist():
notes = ET.fromstring(archive.read("word/footnotes.xml"))
footnotes = sum(
1 for note in notes.iter(f"{_W}footnote") if int(note.get(f"{_W}id", "0")) > 0
)
headings = paragraphs = 0
for para in root.iter(f"{_W}p"):
style = para.find(f"{_W}pPr/{_W}pStyle")
lines = _docx_lines(para)
if style is not None and _HEADING_STYLE.match(style.get(f"{_W}val", "")):
headings += 1
elif _text_of(para, f"{_W}t").strip():
paragraphs += 1
return {
"heading": headings,
"paragraph": paragraphs,
"table": sum(1 for _ in root.iter(f"{_W}tbl")),
"cell": sum(1 for _ in root.iter(f"{_W}tc")),
"image": sum(1 for _ in root.iter(f"{_A}blip")),
"footnote": footnotes,
}
count.add("heading", *lines)
elif lines:
count.add("paragraph", *lines)
for table in root.iter(f"{_W}tbl"):
count.add("table", *[line for p in table.iter(f"{_W}p") for line in _docx_lines(p)])
for cell in root.iter(f"{_W}tc"):
count.add("cell", *[line for p in cell.iter(f"{_W}p") for line in _docx_lines(p)])
for _ in root.iter(f"{_A}blip"):
count.add("image")
if notes is not None:
for note in notes.iter(f"{_W}footnote"):
if int(note.get(f"{_W}id", "0")) > 0:
count.add(
"footnote", *[line for p in note.iter(f"{_W}p") for line in _docx_lines(p)]
)
return count
def count_pptx(data: bytes) -> dict[str, int]:
PPTX = ("cell", "image", "paragraph", "slide", "table", "title")
def count_pptx(data: bytes) -> Count:
"""slide: ppt/slides/slideN.xml. title: a shape whose placeholder is a
title. paragraph: an a:p with text outside a table and outside a title.
table: a:tbl. cell: a:tc. image: p:pic."""
counts = {"slide": 0, "title": 0, "paragraph": 0, "table": 0, "cell": 0, "image": 0}
count = Count(PPTX)
with zipfile.ZipFile(io.BytesIO(data)) as archive:
slides = [n for n in archive.namelist() if re.fullmatch(r"ppt/slides/slide\d+\.xml", n)]
for name in slides:
counts["slide"] += 1
for name in sorted(slides, key=_slide_order):
root = ET.fromstring(archive.read(name))
counts["table"] += sum(1 for _ in root.iter(f"{_A}tbl"))
counts["cell"] += sum(1 for _ in root.iter(f"{_A}tc"))
counts["image"] += sum(1 for _ in root.iter(f"{_P}pic"))
count.add("slide", *_pptx_lines(root))
for table in root.iter(f"{_A}tbl"):
count.add("table", *_pptx_lines(table))
for cell in root.iter(f"{_A}tc"):
count.add("cell", *_pptx_lines(cell))
for _ in root.iter(f"{_P}pic"):
count.add("image")
for shape in root.iter(f"{_P}sp"):
placeholder = shape.find(f"{_P}nvSpPr/{_P}nvPr/{_P}ph")
is_title = placeholder is not None and placeholder.get("type") in (
"title",
"ctrTitle",
)
texts = [p for p in shape.iter(f"{_A}p") if _text_of(p, f"{_A}t").strip()]
texts = [
_normal(_text_of(p, f"{_A}t"))
for p in shape.iter(f"{_A}p")
if _text_of(p, f"{_A}t").strip()
]
if is_title and texts:
counts["title"] += 1
count.add("title", *texts)
else:
counts["paragraph"] += len(texts)
return counts
for text in texts:
count.add("paragraph", text)
return count
def count_xlsx(data: bytes) -> dict[str, int]:
def _pptx_lines(node: ET.Element) -> list[str]:
"""One piece per a:p that holds text."""
return [
_normal(_text_of(p, f"{_A}t")) for p in node.iter(f"{_A}p") if _text_of(p, f"{_A}t").strip()
]
def _slide_order(name: str) -> tuple[int, str]:
match = re.search(r"(\d+)", name)
return (int(match.group(1)) if match else 0, name)
XLSX = ("cell", "image", "row", "sheet")
def _shared_strings(archive: zipfile.ZipFile) -> list[str]:
if "xl/sharedStrings.xml" not in archive.namelist():
return []
root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
return [_normal(_text_of(item, f"{_S}t")) for item in root.iter(f"{_S}si")]
def _cell_text(cell: ET.Element, shared: list[str]) -> str:
inline = cell.find(f"{_S}is")
if inline is not None:
return _normal(_text_of(inline, f"{_S}t"))
value = cell.find(f"{_S}v")
raw = (value.text or "") if value is not None else ""
if cell.get("t") == "s":
try:
return shared[int(raw)]
except (ValueError, IndexError):
return ""
return _normal(raw)
def count_xlsx(data: bytes) -> Count:
"""sheet: xl/worksheets/sheetN.xml. row: a row holding a value. cell: a c
with a value. image: an xdr:pic in a drawing."""
counts = {"sheet": 0, "row": 0, "cell": 0, "image": 0}
with a value. image: an xdr:pic in a drawing.
A cell's text is resolved through `sharedStrings.xml`, because that is
where a spreadsheet's words actually live: the cell holds an index."""
count = Count(XLSX)
with zipfile.ZipFile(io.BytesIO(data)) as archive:
for name in archive.namelist():
shared = _shared_strings(archive)
for name in sorted(archive.namelist()):
if re.fullmatch(r"xl/worksheets/sheet\d+\.xml", name):
counts["sheet"] += 1
root = ET.fromstring(archive.read(name))
sheet_pieces: list[str] = []
for row in root.iter(f"{_S}row"):
valued = [
c
for c in row.iter(f"{_S}c")
if c.find(f"{_S}v") is not None or c.find(f"{_S}is") is not None
]
counts["cell"] += len(valued)
counts["row"] += 1 if valued else 0
values = [_cell_text(c, shared) for c in valued]
for value in values:
count.add("cell", value)
if valued:
count.add("row", *values)
sheet_pieces.extend(values)
count.add("sheet", *sheet_pieces)
elif re.fullmatch(r"xl/drawings/drawing\d+\.xml", name):
root = ET.fromstring(archive.read(name))
counts["image"] += sum(1 for _ in root.iter(f"{_XDR}pic"))
return counts
for _ in root.iter(f"{_XDR}pic"):
count.add("image")
return count
def count_odt(data: bytes) -> dict[str, int]:
ODT = ("cell", "heading", "image", "list_item", "paragraph", "table")
def count_odt(data: bytes) -> Count:
"""heading: text:h. paragraph: a text:p with text outside a table cell.
table: table:table. cell: table:table-cell. list_item: text:list-item.
image: draw:image."""
with zipfile.ZipFile(io.BytesIO(data)) as archive:
root = ET.fromstring(archive.read("content.xml"))
count = Count(ODT)
in_cell: set[int] = set()
for cell in root.iter(f"{_TABLE}table-cell"):
in_cell.update(id(p) for p in cell.iter(f"{_TEXT}p"))
return {
"heading": sum(1 for _ in root.iter(f"{_TEXT}h")),
"paragraph": sum(
1
for p in root.iter(f"{_TEXT}p")
if id(p) not in in_cell and "".join(p.itertext()).strip()
),
"table": sum(1 for _ in root.iter(f"{_TABLE}table")),
"cell": sum(1 for _ in root.iter(f"{_TABLE}table-cell")),
"list_item": sum(1 for _ in root.iter(f"{_TEXT}list-item")),
"image": sum(1 for _ in root.iter(f"{_DRAW}image")),
def lines(node: ET.Element) -> list[str]:
pieces = [_normal("".join(p.itertext())) for p in node.iter(f"{_TEXT}p")]
pieces += [_normal("".join(h.itertext())) for h in node.iter(f"{_TEXT}h")]
return [piece for piece in pieces if piece]
for heading in root.iter(f"{_TEXT}h"):
count.add("heading", _normal("".join(heading.itertext())))
for para in root.iter(f"{_TEXT}p"):
text = _normal("".join(para.itertext()))
if id(para) not in in_cell and text:
count.add("paragraph", text)
for table in root.iter(f"{_TABLE}table"):
count.add("table", *lines(table))
for cell in root.iter(f"{_TABLE}table-cell"):
count.add("cell", *lines(cell))
for item in root.iter(f"{_TEXT}list-item"):
count.add("list_item", *lines(item))
for _ in root.iter(f"{_DRAW}image"):
count.add("image")
return count
RTF = ("cell", "image", "paragraph", "table_row")
_RTF_CONTROL = re.compile(r"\\([a-zA-Z]+)(-?\d+)? ?|\\'([0-9a-fA-F]{2})|\\([^a-zA-Z])")
#: Groups that hold no document text. `{\*...}` says so in the format itself;
#: these say it by name, and without them a fixture's font table reads as the
#: first paragraph of its prose.
_RTF_SILENT = frozenset(
{
"fonttbl",
"colortbl",
"stylesheet",
"info",
"listtable",
"listoverridetable",
"rsidtbl",
"generator",
"filetbl",
"pgptbl",
"themedata",
"colorschememapping",
"latentstyles",
"datastore",
}
)
def count_rtf(text: str) -> dict[str, int]:
"""paragraph: \\par. table_row: \\row. cell: \\cell. image: \\pict.
A control word ends at the first non-letter, so \\pard is not \\par."""
def _rtf_pieces(text: str) -> dict[str, list[str]]:
r"""The text standing before each `\par`, `\cell` and `\row`.
A control word ends at the first non-letter, so `\pard` is not `\par`.
`\uN` carries a character the byte escapes cannot, and the substitution
character standing after it is the same character again -- counted twice,
a Norwegian word reads as `hovedl?pet`.
"""
pieces: dict[str, list[Any]] = {"paragraph": [], "cell": [], "table_row": []}
buffer: list[str] = []
cell: list[str] = []
silent: list[int] = []
depth = 0
position = 0
skip_after_unicode = 0
while position < len(text):
char = text[position]
quiet = bool(silent)
if char == "{":
depth += 1
position += 1
match = re.match(r"\\\*?\\?([a-zA-Z]+)", text[position : position + 32])
if match and match.group(1) in _RTF_SILENT:
silent.append(depth)
elif text[position : position + 2] == "\\*":
silent.append(depth)
continue
if char == "}":
if silent and silent[-1] == depth:
silent.pop()
depth -= 1
position += 1
continue
if char == "\\":
match = _RTF_CONTROL.match(text, position)
if match is None:
position += 1
continue
position = match.end()
word, number, hexcode, symbol = match.groups()
if hexcode is not None:
if not quiet and not skip_after_unicode:
buffer.append(bytes([int(hexcode, 16)]).decode("cp1252", "replace"))
skip_after_unicode = max(0, skip_after_unicode - 1)
continue
if symbol is not None:
continue
if word == "u" and number is not None:
code = int(number)
if not quiet:
buffer.append(chr(code if code >= 0 else code + 65536))
skip_after_unicode = 1
continue
if word == "par":
pieces["paragraph"].append(_normal("".join(buffer)))
buffer = []
elif word == "cell":
joined = _normal("".join(buffer))
pieces["cell"].append(joined)
cell.append(joined)
buffer = []
elif word == "row":
pieces["table_row"].append(cell)
cell = []
continue
if skip_after_unicode and not char.isspace():
skip_after_unicode -= 1
position += 1
continue
if not quiet:
buffer.append(char)
position += 1
return pieces
def count_rtf(text: str) -> Count:
"""paragraph: \\par. table_row: \\row. cell: \\cell. image: \\pict."""
count = Count(RTF)
def word(name: str) -> int:
return len(re.findall(rf"\\{name}(?![a-zA-Z])", text))
return {
"paragraph": word("par"),
"table_row": word("row"),
"cell": word("cell"),
"image": word("pict"),
}
pieces = _rtf_pieces(text)
for role, control in (("paragraph", "par"), ("cell", "cell"), ("table_row", "row")):
found = pieces[role]
for index in range(word(control)):
own = found[index] if index < len(found) else ""
count.add(role, *(own if isinstance(own, list) else [own]))
for _ in range(word("pict")):
count.add("image")
return count
# --- pdf ---------------------------------------------------------------------
def pdf_objects(path: Path) -> dict[str, int] | None:
"""page and image placements as pdfplumber sees them; None without it."""
PDF = ("image", "page")
def _lines(text: str) -> list[str]:
"""One piece per line of a page: a reader re-wraps, and a line survives."""
return [_normal(line) for line in text.split("\n") if line.strip()]
def pdf_objects(path: Path) -> Count | None:
"""page and image placements as pdfplumber sees them; None without it.
A page's text is the page's own text. Until 2026-09-18 this witness saw
a PDF as pages and picture placements alone, so ALL of a PDF's text could
leave the bundle with no row able to see it (independent review, M-1).
"""
try:
import pdfplumber
except ImportError:
return None
count = Count(PDF)
with pdfplumber.open(str(path)) as pdf:
pages = len(pdf.pages)
images = 0
for page in pdf.pages:
images += len(page.images)
for _ in page.images:
count.add("image")
count.add("page", *_lines(page.extract_text() or ""))
page.close()
return {"page": pages, "image": images}
return count
def pdf_poppler(path: Path) -> dict[str, int] | None:
"""page (pdfinfo) and image (pdfimages -list, rows of type `image`); None
when poppler is not installed."""
def pdf_poppler(path: Path) -> Count | None:
"""The same, through poppler (`pdfinfo`, `pdfimages -list`, `pdftotext`);
None when poppler is not installed.
A genuinely independent reader: a different code base, a different text
engine, run as a subprocess."""
info = shutil.which("pdfinfo")
lister = shutil.which("pdfimages")
if info is None or lister is None:
totext = shutil.which("pdftotext")
if info is None or lister is None or totext is None:
return None
meta = subprocess.run([info, str(path)], capture_output=True, text=True, check=True).stdout
pages_match = re.search(r"^Pages:\s+(\d+)", meta, re.MULTILINE)
pages = int(pages_match.group(1)) if pages_match else 0
listing = subprocess.run(
[lister, "-list", str(path)], capture_output=True, text=True, check=True
).stdout
images = 0
count = Count(PDF)
for line in listing.splitlines()[2:]:
cells = line.split()
if len(cells) > 2 and cells[2] == "image":
images += 1
return {"page": int(pages_match.group(1)) if pages_match else 0, "image": images}
count.add("image")
for number in range(1, pages + 1):
page = subprocess.run(
[totext, "-f", str(number), "-l", str(number), str(path), "-"],
capture_output=True,
text=True,
check=True,
).stdout
count.add("page", *_lines(page))
return count
# --- one file ----------------------------------------------------------------
@ -605,47 +966,54 @@ def witness_file(inbox: Path, path: Path) -> Inventory:
refs: list[str] = []
sts = False
if suffix == ".md":
elements, refs = count_markdown(data.decode("utf-8-sig"))
witness = "markdown lines"
count, refs = count_markdown(data.decode("utf-8-sig"))
name = "markdown lines"
elif suffix == ".txt":
elements, witness = count_text(data.decode("utf-8-sig")), "text lines"
count, name = count_text(data.decode("utf-8-sig")), "text lines"
elif suffix == ".csv":
elements, witness = count_csv(data.decode("utf-8-sig")), "csv"
count, name = count_csv(data.decode("utf-8-sig")), "csv"
elif suffix == ".json":
elements, witness = count_json(data.decode("utf-8-sig")), "json"
count, name = count_json(data.decode("utf-8-sig")), "json"
elif suffix in (".html", ".htm"):
parser = _HtmlCounter()
parser.feed(data.decode("utf-8-sig"))
parser.close()
elements, refs, witness = parser.elements, parser.refs, "html.parser"
parser.finish()
count, refs, name = parser.count, parser.refs, "html.parser"
elif suffix == ".xml":
elements, refs, sts = count_sts_xml(data)
witness = "xml.etree"
count, refs, sts = count_sts_xml(data)
name = "xml.etree"
elif suffix == ".docx":
elements, witness = count_docx(data), "docx zip xml"
count, name = count_docx(data), "docx zip xml"
elif suffix == ".pptx":
elements, witness = count_pptx(data), "pptx zip xml"
count, name = count_pptx(data), "pptx zip xml"
elif suffix == ".xlsx":
elements, witness = count_xlsx(data), "xlsx zip xml"
count, name = count_xlsx(data), "xlsx zip xml"
elif suffix == ".odt":
elements, witness = count_odt(data), "odt zip xml"
count, name = count_odt(data), "odt zip xml"
elif suffix == ".rtf":
elements, witness = count_rtf(data.decode("latin-1")), "rtf control words"
count, name = count_rtf(data.decode("latin-1")), "rtf control words"
elif suffix == ".pdf":
objects = pdf_objects(path)
if objects is None:
raise WitnessRefused("pdfplumber is not installed")
elements, witness = objects, "pdfplumber objects"
count, name = objects, "pdfplumber objects"
else:
raise WitnessRefused(f"no witness for {suffix or 'a file without a suffix'}")
inventory = Inventory(relative, suffix, witness, dict(elements))
inventory = Inventory(
relative,
suffix,
name,
dict(count.counts),
texts={role: [list(p) for p in values] for role, values in count.texts.items()},
)
for ref in refs:
if not ref or _is_remote(ref):
inventory.images.append(ImageRef(REMOTE, ref))
else:
target = resolve_local(inbox, path, ref, sts=sts)
inventory.images.append(ImageRef(LOCAL, ref, target))
embedded = elements.get("image", 0) - len(refs)
embedded = count.counts.get("image", 0) - len(refs)
inventory.images.extend(ImageRef(EMBEDDED, "") for _ in range(max(0, embedded)))
return inventory