Two MAJOR findings of the independent v0.10.0 review, both with the shipped defaults, both new in 0.10.0. Repros rebuilt as tests first. - A remote <img src>/xlink:href became a LIVE markdown image link in the persisted concept, with the address and query string chosen by whoever wrote the document. Extraction opens no socket; a consumer rendering the bundle does. Now inert text with the address in a code span, pinned by a property over the readers rather than by one string. The tier asymmetry (user-upload refuses, trusted-source persisted) went to the guard repo with the repro. - Nothing bounded a declared image size: 9.6 KB of PDF declaring 3000x3000 grayscale zeros took 83 MB peak RSS, linear in pixels. MAX_IMAGE_PIXELS (40 000 000) and MAX_IMAGE_BYTES (256 MiB) are read off the corpora (largest measured 18.6 MP on K2, 1.4 MP on R761) and checked on what the container declares, before any decompression; over them is asset_too_large, counted. The same bound closes the inline data: URI, which the review flagged and did not measure. Also fixed, added by PM to this order: an inline PDF image was named from id() of a Python object, so two concept files of the reference corpus differed between builds. It is now named from its position. R761 unchanged: 50 carried of 50 found, assets diff -rq clean. Report: docs/2026-09-17-bildestien-0-10-1.md Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
249 lines
11 KiB
Python
249 lines
11 KiB
Python
"""Two findings of the independent 0.10.0 review, as red tests (0.10.1).
|
|
|
|
Both land with the SHIPPED defaults (`--assets` on, `--gate
|
|
guard-trusted-source`), and both are new in 0.10.0 -- before it, no reader
|
|
read an `<img>` attribute or opened an image stream at all.
|
|
|
|
1. **A remote image reference became a LIVE markdown image link** in the
|
|
persisted concept, carrying an address the document's author controls,
|
|
query string included. This package opens no socket, but a consumer that
|
|
renders the markdown or lets an agent fetch images does, which turns "a
|
|
bundle was opened" into a beacon. The guard's `user-upload` tier refuses
|
|
such a line and the build's default tier does not, so the same bytes are
|
|
persisted under the default and refused one tier up.
|
|
2. **Nothing bounded a PDF image's size.** A 9.6 KB file declaring an
|
|
8000x8000 grayscale image of compressed zeros made `encode_png` allocate
|
|
width*height bytes twice over; measured peak RSS 83 MB at 3000x3000 and
|
|
276 MB at 8000x8000, linear in the pixel count. One such document -- or
|
|
one legitimately enormous scan -- takes the whole batch build with it,
|
|
before any gate, because the guard never sees image bytes.
|
|
|
|
The limit is READ OFF the corpora rather than chosen: over the 4 828 image
|
|
objects of the 43-document reference corpus the largest is 4 515 x 4 128
|
|
(18.6 MP, a landscape drawing), and over R761's 109 delivered pictures the
|
|
largest is 2 072 x 656 (1.4 MP). `MAX_IMAGE_PIXELS` sits above both with
|
|
room to spare, and anything larger is a counted refusal rather than a
|
|
killed build.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import zlib
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf import assets, cli
|
|
from llm_ingestion_okf.assets import IMAGE_POINTER, render_missing
|
|
from llm_ingestion_okf.errors import ExtractionError
|
|
from llm_ingestion_okf.extract import extract_document
|
|
|
|
FIXTURES = Path(__file__).parent / "fixtures" / "image-inbox"
|
|
REMOTE = "https://collect.example.net/p.gif?doc=drift&u=SESSION"
|
|
|
|
#: Any markdown image whose target is not this bundle's own `assets/`.
|
|
FOREIGN_IMAGE_LINK = re.compile(r"!\[[^\]\n]*\]\(\s*(?!/assets/)([^)\s]+)")
|
|
|
|
|
|
# --- finding 1: a remote reference is inert ---------------------------------
|
|
|
|
|
|
def test_a_remote_reference_is_not_a_markdown_image_link() -> None:
|
|
line = render_missing(REMOTE, reason="the source is off this machine", label="fig", href=REMOTE)
|
|
assert "](" not in line
|
|
assert REMOTE in line, "the address is still stated -- a reader must see what was there"
|
|
|
|
|
|
def test_the_pointer_block_of_a_carried_image_is_unchanged() -> None:
|
|
"""The known-positive beside it: a local image keeps its image block, or
|
|
the fix has merely disarmed the whole capability."""
|
|
image = assets.read_image((FIXTURES / "graphics" / "figur-84-1.png").read_bytes(), name="f.png")
|
|
assert IMAGE_POINTER.search(assets.render_block(image)) is not None
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"document",
|
|
[
|
|
'<html><body><p>A</p><img src="{ref}" alt="fig"></body></html>',
|
|
'<standard xmlns:xlink="http://www.w3.org/1999/xlink"><body><sec><title>T</title>'
|
|
'<p>A</p><graphic xlink:href="{ref}"/></sec></body></standard>',
|
|
],
|
|
ids=["html", "sts"],
|
|
)
|
|
@pytest.mark.parametrize("ref", [REMOTE, "//collect.example.net/p.gif", "HTTPS://EVIL/p.gif"])
|
|
def test_no_reader_writes_a_live_link_for_a_remote_reference(document: str, ref: str) -> None:
|
|
"""A property over the readers that resolve references, not one string."""
|
|
suffix = ".html" if document.startswith("<html") else ".xml"
|
|
# `&` is an entity opener in XML, so the reference is escaped for that
|
|
# reader and not for the HTML one.
|
|
escaped = ref if suffix == ".html" else ref.replace("&", "&")
|
|
extracted = extract_document(
|
|
f"doc{suffix}", document.format(ref=escaped).encode("utf-8"), assets=True
|
|
)
|
|
foreign = FOREIGN_IMAGE_LINK.findall(extracted.text)
|
|
assert foreign == [], f"live image link(s) {foreign} for {ref}"
|
|
assert ref.split("?")[0].lower() in extracted.text.lower()
|
|
|
|
|
|
def test_a_data_uri_image_is_carried_and_leaves_no_foreign_link() -> None:
|
|
png = (FIXTURES / "graphics" / "figur-84-1.png").read_bytes()
|
|
import base64
|
|
|
|
uri = "data:image/png;base64," + base64.b64encode(png).decode("ascii")
|
|
extracted = extract_document(
|
|
"doc.html", f'<html><body><img src="{uri}"></body></html>'.encode(), assets=True
|
|
)
|
|
assert len(extracted.images) == 1
|
|
assert FOREIGN_IMAGE_LINK.findall(extracted.text) == []
|
|
|
|
|
|
def test_a_built_bundle_carries_no_foreign_image_link(tmp_path: Path) -> None:
|
|
"""The shipped fixture inbox holds one remote `<img>`; the bundle must
|
|
point at `assets/` and nowhere else."""
|
|
pytest.importorskip("pdfplumber")
|
|
pytest.importorskip("pypandoc")
|
|
bundle = tmp_path / "bundle"
|
|
code = cli.main(
|
|
[
|
|
"build",
|
|
str(FIXTURES),
|
|
"--bundle",
|
|
str(bundle),
|
|
"--bundle-id",
|
|
"limits",
|
|
"--okf-version",
|
|
"0.2",
|
|
]
|
|
)
|
|
assert code == 0
|
|
foreign: list[str] = []
|
|
for path in bundle.rglob("*.md"):
|
|
foreign += FOREIGN_IMAGE_LINK.findall(path.read_text(encoding="utf-8"))
|
|
assert foreign == []
|
|
|
|
|
|
# --- finding 2: a declared size that is too large is refused, not decoded ---
|
|
|
|
|
|
def _bomb(dimension: int, content: bytes | None = None) -> bytes:
|
|
"""A tiny PDF declaring one `dimension` x `dimension` grayscale image of
|
|
compressed zeros -- the review's own repro, built here. `content` replaces
|
|
the page's content stream, for a page that draws an INLINE image instead."""
|
|
payload = zlib.compress(b"\x00" * (dimension * dimension), 9)
|
|
if content is None:
|
|
content = b"BT /F1 12 Tf 20 100 Td (bomb) Tj ET\nq 100 0 0 100 20 20 cm /Im0 Do Q\n"
|
|
objects = [
|
|
b"<< /Type /Catalog /Pages 2 0 R >>",
|
|
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
|
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /XObject "
|
|
b"<< /Im0 5 0 R >> /Font << /F1 6 0 R >> >> /Contents 4 0 R >>",
|
|
b"<< /Length %d >>\nstream\n" % len(content) + content + b"\nendstream",
|
|
(
|
|
"<< /Type /XObject /Subtype /Image /Width %d /Height %d /ColorSpace /DeviceGray "
|
|
"/BitsPerComponent 8 /Filter /FlateDecode /Length %d >>\nstream\n"
|
|
% (dimension, dimension, len(payload))
|
|
).encode("ascii")
|
|
+ payload
|
|
+ b"\nendstream",
|
|
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
|
]
|
|
out = bytearray(b"%PDF-1.4\n")
|
|
offsets = []
|
|
for number, body in enumerate(objects, start=1):
|
|
offsets.append(len(out))
|
|
out += b"%d 0 obj\n" % number + body + b"\nendobj\n"
|
|
start = len(out)
|
|
out += b"xref\n0 %d\n0000000000 65535 f \n" % (len(objects) + 1)
|
|
for offset in offsets:
|
|
out += b"%010d 00000 n \n" % offset
|
|
out += b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n" % (
|
|
len(objects) + 1,
|
|
start,
|
|
)
|
|
return bytes(out)
|
|
|
|
|
|
def test_the_limit_is_above_every_image_measured_in_the_corpora() -> None:
|
|
"""4 515 x 4 128 = 18.6 MP is the largest of the 4 828 objects measured in
|
|
the reference corpus; R761's largest delivered picture is 1.4 MP."""
|
|
assert assets.MAX_IMAGE_PIXELS > 4515 * 4128
|
|
assert assets.MAX_IMAGE_BYTES >= assets.MAX_IMAGE_PIXELS
|
|
|
|
|
|
def test_a_pdf_image_over_the_limit_is_refused_by_code() -> None:
|
|
pytest.importorskip("pdfplumber")
|
|
dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5)
|
|
extracted = extract_document("bomb.pdf", _bomb(dimension), assets=True)
|
|
assert extracted.images == ()
|
|
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
|
|
assert str(dimension) in extracted.rejected[0].reason
|
|
|
|
|
|
def test_a_pdf_image_under_the_limit_is_still_carried() -> None:
|
|
"""The boundary from the other side, on the same generator."""
|
|
pytest.importorskip("pdfplumber")
|
|
extracted = extract_document("small.pdf", _bomb(64), assets=True)
|
|
assert [(image.width, image.height) for image in extracted.images] == [(64, 64)]
|
|
|
|
|
|
def test_the_refusal_happens_before_the_stream_is_decompressed() -> None:
|
|
"""The limit is read off the DECLARED size, and the order is observable.
|
|
|
|
The image stream here is CORRUPT (its bytes are not deflate data) while
|
|
its declared size is over the bound. Decoding first gives
|
|
`asset_pdf_unsupported` ("could not be decoded"); reading the declared
|
|
size first gives `asset_too_large`. A check after `get_data()` has already
|
|
paid for the bomb it was meant to stop.
|
|
"""
|
|
pytest.importorskip("pdfplumber")
|
|
dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5)
|
|
document = _bomb(dimension)
|
|
payload = zlib.compress(b"\x00" * (dimension * dimension), 9)
|
|
corrupt = document.replace(payload, b"\xff" * len(payload))
|
|
assert corrupt != document
|
|
extracted = extract_document("corrupt.pdf", corrupt, assets=True)
|
|
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
|
|
|
|
|
|
def test_a_data_uri_over_the_limit_is_refused_before_decoding(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
monkeypatch.setattr(assets, "MAX_IMAGE_BYTES", 32)
|
|
uri = "data:image/png;base64," + "A" * 4096
|
|
extracted = extract_document(
|
|
"doc.html", f'<html><body><img src="{uri}"></body></html>'.encode(), assets=True
|
|
)
|
|
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
|
|
|
|
|
|
def test_encode_png_refuses_the_same_size_on_its_own() -> None:
|
|
"""Defence in depth: the encoder does not trust its caller to have checked."""
|
|
dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5)
|
|
with pytest.raises(ExtractionError) as excinfo:
|
|
assets.encode_png(dimension, dimension, b"", channels=1, palette=None, alpha=None)
|
|
assert excinfo.value.code == "asset_too_large"
|
|
|
|
|
|
# --- the determinism defect PM added to this order --------------------------
|
|
|
|
|
|
def test_an_inline_pdf_image_gets_a_stable_name() -> None:
|
|
"""pdfminer names an inline image (`BI ... EI`) from `id()` of a Python
|
|
object, so the pointer line changed between two runs of one build -- two
|
|
concept files of the reference corpus differed. Measured 2026-09-17."""
|
|
pytest.importorskip("pdfplumber")
|
|
from llm_ingestion_okf import extract as extract_module
|
|
|
|
inline = (
|
|
b"BT /F1 12 Tf 20 100 Td (t) Tj ET\n"
|
|
b"q 10 0 0 10 20 20 cm BI /W 2 /H 2 /CS /G /BPC 8 /F /AHx ID 00112233> EI Q\n"
|
|
)
|
|
document = _bomb(4, content=inline)
|
|
first = extract_document("inline.pdf", document, assets=True)
|
|
extract_module._pdf_pages.cache_clear()
|
|
second = extract_document("inline.pdf", document, assets=True)
|
|
names = [image.name for image in first.images] + [r.name for r in first.rejected]
|
|
again = [image.name for image in second.images] + [r.name for r in second.rejected]
|
|
assert names == again != []
|
|
assert not any(part.isdigit() and len(part) > 6 for name in names for part in name.split("-"))
|