"""Two findings of the independent 0.10.0 review, as red tests (0.10.1).
Both land with the SHIPPED defaults (`--assets` on, `--gate
guard-trusted-source`), and both are new in 0.10.0 -- before it, no reader
read an `
` attribute or opened an image stream at all.
1. **A remote image reference became a LIVE markdown image link** in the
persisted concept, carrying an address the document's author controls,
query string included. This package opens no socket, but a consumer that
renders the markdown or lets an agent fetch images does, which turns "a
bundle was opened" into a beacon. The guard's `user-upload` tier refuses
such a line and the build's default tier does not, so the same bytes are
persisted under the default and refused one tier up.
2. **Nothing bounded a PDF image's size.** A 9.6 KB file declaring an
8000x8000 grayscale image of compressed zeros made `encode_png` allocate
width*height bytes twice over; measured peak RSS 83 MB at 3000x3000 and
276 MB at 8000x8000, linear in the pixel count. One such document -- or
one legitimately enormous scan -- takes the whole batch build with it,
before any gate, because the guard never sees image bytes.
The limit is READ OFF the corpora rather than chosen: over the 4 828 image
objects of the 43-document reference corpus the largest is 4 515 x 4 128
(18.6 MP, a landscape drawing), and over R761's 109 delivered pictures the
largest is 2 072 x 656 (1.4 MP). `MAX_IMAGE_PIXELS` sits above both with
room to spare, and anything larger is a counted refusal rather than a
killed build.
"""
from __future__ import annotations
import re
import zlib
from pathlib import Path
import pytest
from llm_ingestion_okf import assets, cli
from llm_ingestion_okf.assets import IMAGE_POINTER, render_missing
from llm_ingestion_okf.errors import ExtractionError
from llm_ingestion_okf.extract import extract_document
FIXTURES = Path(__file__).parent / "fixtures" / "image-inbox"
REMOTE = "https://collect.example.net/p.gif?doc=drift&u=SESSION"
#: Any markdown image whose target is not this bundle's own `assets/`.
FOREIGN_IMAGE_LINK = re.compile(r"!\[[^\]\n]*\]\(\s*(?!/assets/)([^)\s]+)")
# --- finding 1: a remote reference is inert ---------------------------------
def test_a_remote_reference_is_not_a_markdown_image_link() -> None:
line = render_missing(REMOTE, reason="the source is off this machine", label="fig", href=REMOTE)
assert "](" not in line
assert REMOTE in line, "the address is still stated -- a reader must see what was there"
def test_the_pointer_block_of_a_carried_image_is_unchanged() -> None:
"""The known-positive beside it: a local image keeps its image block, or
the fix has merely disarmed the whole capability."""
image = assets.read_image((FIXTURES / "graphics" / "figur-84-1.png").read_bytes(), name="f.png")
assert IMAGE_POINTER.search(assets.render_block(image)) is not None
@pytest.mark.parametrize(
"document",
[
'
A
',
'T'
'A
',
],
ids=["html", "sts"],
)
@pytest.mark.parametrize("ref", [REMOTE, "//collect.example.net/p.gif", "HTTPS://EVIL/p.gif"])
def test_no_reader_writes_a_live_link_for_a_remote_reference(document: str, ref: str) -> None:
"""A property over the readers that resolve references, not one string."""
suffix = ".html" if document.startswith(" None:
png = (FIXTURES / "graphics" / "figur-84-1.png").read_bytes()
import base64
uri = "data:image/png;base64," + base64.b64encode(png).decode("ascii")
extracted = extract_document(
"doc.html", f'
'.encode(), assets=True
)
assert len(extracted.images) == 1
assert FOREIGN_IMAGE_LINK.findall(extracted.text) == []
def test_a_built_bundle_carries_no_foreign_image_link(tmp_path: Path) -> None:
"""The shipped fixture inbox holds one remote `
`; the bundle must
point at `assets/` and nowhere else."""
pytest.importorskip("pdfplumber")
pytest.importorskip("pypandoc")
bundle = tmp_path / "bundle"
code = cli.main(
[
"build",
str(FIXTURES),
"--bundle",
str(bundle),
"--bundle-id",
"limits",
"--okf-version",
"0.2",
]
)
assert code == 0
foreign: list[str] = []
for path in bundle.rglob("*.md"):
foreign += FOREIGN_IMAGE_LINK.findall(path.read_text(encoding="utf-8"))
assert foreign == []
# --- finding 2: a declared size that is too large is refused, not decoded ---
def _bomb(dimension: int, content: bytes | None = None) -> bytes:
"""A tiny PDF declaring one `dimension` x `dimension` grayscale image of
compressed zeros -- the review's own repro, built here. `content` replaces
the page's content stream, for a page that draws an INLINE image instead."""
payload = zlib.compress(b"\x00" * (dimension * dimension), 9)
if content is None:
content = b"BT /F1 12 Tf 20 100 Td (bomb) Tj ET\nq 100 0 0 100 20 20 cm /Im0 Do Q\n"
objects = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /XObject "
b"<< /Im0 5 0 R >> /Font << /F1 6 0 R >> >> /Contents 4 0 R >>",
b"<< /Length %d >>\nstream\n" % len(content) + content + b"\nendstream",
(
"<< /Type /XObject /Subtype /Image /Width %d /Height %d /ColorSpace /DeviceGray "
"/BitsPerComponent 8 /Filter /FlateDecode /Length %d >>\nstream\n"
% (dimension, dimension, len(payload))
).encode("ascii")
+ payload
+ b"\nendstream",
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
]
out = bytearray(b"%PDF-1.4\n")
offsets = []
for number, body in enumerate(objects, start=1):
offsets.append(len(out))
out += b"%d 0 obj\n" % number + body + b"\nendobj\n"
start = len(out)
out += b"xref\n0 %d\n0000000000 65535 f \n" % (len(objects) + 1)
for offset in offsets:
out += b"%010d 00000 n \n" % offset
out += b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n" % (
len(objects) + 1,
start,
)
return bytes(out)
def test_the_limit_is_above_every_image_measured_in_the_corpora() -> None:
"""4 515 x 4 128 = 18.6 MP is the largest of the 4 828 objects measured in
the reference corpus; R761's largest delivered picture is 1.4 MP."""
assert assets.MAX_IMAGE_PIXELS > 4515 * 4128
assert assets.MAX_IMAGE_BYTES >= assets.MAX_IMAGE_PIXELS
def test_a_pdf_image_over_the_limit_is_refused_by_code() -> None:
pytest.importorskip("pdfplumber")
dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5)
extracted = extract_document("bomb.pdf", _bomb(dimension), assets=True)
assert extracted.images == ()
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
assert str(dimension) in extracted.rejected[0].reason
def test_a_pdf_image_under_the_limit_is_still_carried() -> None:
"""The boundary from the other side, on the same generator."""
pytest.importorskip("pdfplumber")
extracted = extract_document("small.pdf", _bomb(64), assets=True)
assert [(image.width, image.height) for image in extracted.images] == [(64, 64)]
def test_the_refusal_happens_before_the_stream_is_decompressed() -> None:
"""The limit is read off the DECLARED size, and the order is observable.
The image stream here is CORRUPT (its bytes are not deflate data) while
its declared size is over the bound. Decoding first gives
`asset_pdf_unsupported` ("could not be decoded"); reading the declared
size first gives `asset_too_large`. A check after `get_data()` has already
paid for the bomb it was meant to stop.
"""
pytest.importorskip("pdfplumber")
dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5)
document = _bomb(dimension)
payload = zlib.compress(b"\x00" * (dimension * dimension), 9)
corrupt = document.replace(payload, b"\xff" * len(payload))
assert corrupt != document
extracted = extract_document("corrupt.pdf", corrupt, assets=True)
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
def test_a_data_uri_over_the_limit_is_refused_before_decoding(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(assets, "MAX_IMAGE_BYTES", 32)
uri = "data:image/png;base64," + "A" * 4096
extracted = extract_document(
"doc.html", f'
'.encode(), assets=True
)
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
def test_encode_png_refuses_the_same_size_on_its_own() -> None:
"""Defence in depth: the encoder does not trust its caller to have checked."""
dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5)
with pytest.raises(ExtractionError) as excinfo:
assets.encode_png(dimension, dimension, b"", channels=1, palette=None, alpha=None)
assert excinfo.value.code == "asset_too_large"
# --- the determinism defect PM added to this order --------------------------
def test_an_inline_pdf_image_gets_a_stable_name() -> None:
"""pdfminer names an inline image (`BI ... EI`) from `id()` of a Python
object, so the pointer line changed between two runs of one build -- two
concept files of the reference corpus differed. Measured 2026-09-17."""
pytest.importorskip("pdfplumber")
from llm_ingestion_okf import extract as extract_module
inline = (
b"BT /F1 12 Tf 20 100 Td (t) Tj ET\n"
b"q 10 0 0 10 20 20 cm BI /W 2 /H 2 /CS /G /BPC 8 /F /AHx ID 00112233> EI Q\n"
)
document = _bomb(4, content=inline)
first = extract_document("inline.pdf", document, assets=True)
extract_module._pdf_pages.cache_clear()
second = extract_document("inline.pdf", document, assets=True)
names = [image.name for image in first.images] + [r.name for r in first.rejected]
again = [image.name for image in second.images] + [r.name for r in second.rejected]
assert names == again != []
assert not any(part.isdigit() and len(part) > 6 for name in names for part in name.split("-"))