"""Two findings of the independent 0.10.0 review, as red tests (0.10.1). Both land with the SHIPPED defaults (`--assets` on, `--gate guard-trusted-source`), and both are new in 0.10.0 -- before it, no reader read an `` attribute or opened an image stream at all. 1. **A remote image reference became a LIVE markdown image link** in the persisted concept, carrying an address the document's author controls, query string included. This package opens no socket, but a consumer that renders the markdown or lets an agent fetch images does, which turns "a bundle was opened" into a beacon. The guard's `user-upload` tier refuses such a line and the build's default tier does not, so the same bytes are persisted under the default and refused one tier up. 2. **Nothing bounded a PDF image's size.** A 9.6 KB file declaring an 8000x8000 grayscale image of compressed zeros made `encode_png` allocate width*height bytes twice over; measured peak RSS 83 MB at 3000x3000 and 276 MB at 8000x8000, linear in the pixel count. One such document -- or one legitimately enormous scan -- takes the whole batch build with it, before any gate, because the guard never sees image bytes. The limit is READ OFF the corpora rather than chosen: over the 4 828 image objects of the 43-document reference corpus the largest is 4 515 x 4 128 (18.6 MP, a landscape drawing), and over R761's 109 delivered pictures the largest is 2 072 x 656 (1.4 MP). `MAX_IMAGE_PIXELS` sits above both with room to spare, and anything larger is a counted refusal rather than a killed build. """ from __future__ import annotations import re import subprocess import sys import zlib from pathlib import Path import pytest from llm_ingestion_okf import assets, cli from llm_ingestion_okf.assets import IMAGE_POINTER, render_missing from llm_ingestion_okf.errors import ExtractionError from llm_ingestion_okf.extract import extract_document FIXTURES = Path(__file__).parent / "fixtures" / "image-inbox" REMOTE = "https://collect.example.net/p.gif?doc=drift&u=SESSION" #: Any markdown image whose target is not this bundle's own `assets/`. FOREIGN_IMAGE_LINK = re.compile(r"!\[[^\]\n]*\]\(\s*(?!/assets/)([^)\s]+)") # --- finding 1: a remote reference is inert --------------------------------- def test_a_remote_reference_is_not_a_markdown_image_link() -> None: line = render_missing(REMOTE, reason="the source is off this machine", label="fig", href=REMOTE) assert "](" not in line assert REMOTE in line, "the address is still stated -- a reader must see what was there" def test_the_pointer_block_of_a_carried_image_is_unchanged() -> None: """The known-positive beside it: a local image keeps its image block, or the fix has merely disarmed the whole capability.""" image = assets.read_image((FIXTURES / "graphics" / "figur-84-1.png").read_bytes(), name="f.png") assert IMAGE_POINTER.search(assets.render_block(image)) is not None @pytest.mark.parametrize( "document", [ '

A

fig', 'T' '

A

', ], ids=["html", "sts"], ) @pytest.mark.parametrize("ref", [REMOTE, "//collect.example.net/p.gif", "HTTPS://EVIL/p.gif"]) def test_no_reader_writes_a_live_link_for_a_remote_reference(document: str, ref: str) -> None: """A property over the readers that resolve references, not one string.""" suffix = ".html" if document.startswith(" None: png = (FIXTURES / "graphics" / "figur-84-1.png").read_bytes() import base64 uri = "data:image/png;base64," + base64.b64encode(png).decode("ascii") extracted = extract_document( "doc.html", f''.encode(), assets=True ) assert len(extracted.images) == 1 assert FOREIGN_IMAGE_LINK.findall(extracted.text) == [] def test_a_built_bundle_carries_no_foreign_image_link(tmp_path: Path) -> None: """The shipped fixture inbox holds one remote ``; the bundle must point at `assets/` and nowhere else.""" pytest.importorskip("pdfplumber") pytest.importorskip("pypandoc") bundle = tmp_path / "bundle" code = cli.main( [ "build", str(FIXTURES), "--bundle", str(bundle), "--bundle-id", "limits", "--okf-version", "0.2", ] ) assert code == 0 foreign: list[str] = [] for path in bundle.rglob("*.md"): foreign += FOREIGN_IMAGE_LINK.findall(path.read_text(encoding="utf-8")) assert foreign == [] # --- finding 2: a declared size that is too large is refused, not decoded --- def _zeros_stream(total: int) -> bytes: """`total` bytes of zeros, deflated WITHOUT ever holding them. The generator has to stay cheaper than the bomb it builds, or the test measures its own fixture instead of the code under test. """ compressor = zlib.compressobj(9) chunk = b"\x00" * (1 << 20) parts = [compressor.compress(chunk) for _ in range(total >> 20)] parts.append(compressor.flush()) return b"".join(parts) def _bomb(dimension: int, content: bytes | None = None, payload: bytes | None = None) -> bytes: """A tiny PDF declaring one `dimension` x `dimension` grayscale image of compressed zeros -- the review's own repro, built here. `content` replaces the page's content stream, for a page that draws an INLINE image instead. `payload` replaces the image stream, for a document whose DECLARED size and whose actual stream are two different numbers.""" if payload is None: payload = zlib.compress(b"\x00" * (dimension * dimension), 9) if content is None: content = b"BT /F1 12 Tf 20 100 Td (bomb) Tj ET\nq 100 0 0 100 20 20 cm /Im0 Do Q\n" objects = [ b"<< /Type /Catalog /Pages 2 0 R >>", b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Resources << /XObject " b"<< /Im0 5 0 R >> /Font << /F1 6 0 R >> >> /Contents 4 0 R >>", b"<< /Length %d >>\nstream\n" % len(content) + content + b"\nendstream", ( "<< /Type /XObject /Subtype /Image /Width %d /Height %d /ColorSpace /DeviceGray " "/BitsPerComponent 8 /Filter /FlateDecode /Length %d >>\nstream\n" % (dimension, dimension, len(payload)) ).encode("ascii") + payload + b"\nendstream", b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", ] out = bytearray(b"%PDF-1.4\n") offsets = [] for number, body in enumerate(objects, start=1): offsets.append(len(out)) out += b"%d 0 obj\n" % number + body + b"\nendobj\n" start = len(out) out += b"xref\n0 %d\n0000000000 65535 f \n" % (len(objects) + 1) for offset in offsets: out += b"%010d 00000 n \n" % offset out += b"trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n" % ( len(objects) + 1, start, ) return bytes(out) def test_the_limit_is_above_every_image_measured_in_the_corpora() -> None: """4 515 x 4 128 = 18.6 MP is the largest of the 4 828 objects measured in the reference corpus; R761's largest delivered picture is 1.4 MP.""" assert assets.MAX_IMAGE_PIXELS > 4515 * 4128 assert assets.MAX_IMAGE_BYTES >= assets.MAX_IMAGE_PIXELS def test_a_pdf_image_over_the_limit_is_refused_by_code() -> None: pytest.importorskip("pdfplumber") dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5) extracted = extract_document("bomb.pdf", _bomb(dimension), assets=True) assert extracted.images == () assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"] assert str(dimension) in extracted.rejected[0].reason def test_a_pdf_image_under_the_limit_is_still_carried() -> None: """The boundary from the other side, on the same generator.""" pytest.importorskip("pdfplumber") extracted = extract_document("small.pdf", _bomb(64), assets=True) assert [(image.width, image.height) for image in extracted.images] == [(64, 64)] def test_the_refusal_happens_before_the_stream_is_decompressed() -> None: """The limit is read off the DECLARED size, and the order is observable. The image stream here is CORRUPT (its bytes are not deflate data) while its declared size is over the bound. Decoding first gives `asset_pdf_unsupported` ("could not be decoded"); reading the declared size first gives `asset_too_large`. A check after `get_data()` has already paid for the bomb it was meant to stop. """ pytest.importorskip("pdfplumber") dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5) document = _bomb(dimension) payload = zlib.compress(b"\x00" * (dimension * dimension), 9) corrupt = document.replace(payload, b"\xff" * len(payload)) assert corrupt != document extracted = extract_document("corrupt.pdf", corrupt, assets=True) assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"] def test_a_data_uri_over_the_limit_is_refused_before_decoding( monkeypatch: pytest.MonkeyPatch, ) -> None: monkeypatch.setattr(assets, "MAX_IMAGE_BYTES", 32) uri = "data:image/png;base64," + "A" * 4096 extracted = extract_document( "doc.html", f''.encode(), assets=True ) assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"] def test_encode_png_refuses_the_same_size_on_its_own() -> None: """Defence in depth: the encoder does not trust its caller to have checked.""" dimension = 1 + int(assets.MAX_IMAGE_PIXELS**0.5) with pytest.raises(ExtractionError) as excinfo: assets.encode_png(dimension, dimension, b"", channels=1, palette=None, alpha=None) assert excinfo.value.code == "asset_too_large" # --- the determinism defect PM added to this order -------------------------- def test_an_inline_pdf_image_gets_a_stable_name() -> None: """pdfminer names an inline image (`BI ... EI`) from `id()` of a Python object, so the pointer line changed between two runs of one build -- two concept files of the reference corpus differed. Measured 2026-09-17.""" pytest.importorskip("pdfplumber") from llm_ingestion_okf import extract as extract_module inline = ( b"BT /F1 12 Tf 20 100 Td (t) Tj ET\n" b"q 10 0 0 10 20 20 cm BI /W 2 /H 2 /CS /G /BPC 8 /F /AHx ID 00112233> EI Q\n" ) document = _bomb(4, content=inline) first = extract_document("inline.pdf", document, assets=True) extract_module._pdf_pages.cache_clear() second = extract_document("inline.pdf", document, assets=True) names = [image.name for image in first.images] + [r.name for r in first.rejected] again = [image.name for image in second.images] + [r.name for r in second.rejected] assert names == again != [] assert not any(part.isdigit() and len(part) > 6 for name in names for part in name.split("-")) # --- BLOCKER-1 of the 18.09 review: the bound must bind what the run PAYS ---- # # `check_size` reads `/Width` and `/Height` out of the image dictionary, which # is a CLAIM by an untrusted document, and the claim and the cost are two # independent numbers: `/Length` is the COMPRESSED length and nothing in the # dictionary states what `get_data()` will return. Measured on `230d1cb` by an # independent review: a 389 626-byte PDF declaring 1x1 and carrying 400 MB of # deflated zeros was CARRIED, with no rejection, at 834 MB of peak RSS -- and # 1,2 GB of zeros at 2 436 MB, linear, about 2 100x the file size. The four # mutations that suite already kills do not separate declared from actual, so # they were all green while this stood. #: What a bounded run of the 400 MB bomb may cost, in bytes of peak RSS. #: Measured 2026-09-18 on this machine, same commit, same fixture: 892 MB #: without the bound and 54 MB with it, and the bounded figure barely moves #: when the stream triples (62 MB at 1,2 GB) because what grows is the #: COMPRESSED input, which was already in memory. The bar sits between the #: two, far enough above the bounded run that the interpreter's own footprint #: on another machine cannot reach it. PEAK_RSS_BOUND = 256 * 1024 * 1024 #: The stream the bomb inflates to. Over `MAX_IMAGE_BYTES` (256 MiB), so it is #: refused at the real bound rather than at a monkeypatched one. BOMB_STREAM_BYTES = 400 * 1024 * 1024 _CHILD = """ import resource, sys sys.path.insert(0, {tests!r}) from test_asset_limits import _bomb, _zeros_stream from llm_ingestion_okf.extract import extract_document document = _bomb(1, payload=_zeros_stream({total})) extracted = extract_document("bomb.pdf", document, assets=True) peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss print( len(document), len(extracted.images), ",".join(rejection.code for rejection in extracted.rejected) or "-", peak if sys.platform == "darwin" else peak * 1024, ) """ def _run_bomb(total: int) -> tuple[int, int, str, int]: """The bomb in its own interpreter, so peak RSS is ITS peak and not the high-water mark of every test that ran before it.""" completed = subprocess.run( [sys.executable, "-c", _CHILD.format(tests=str(Path(__file__).parent), total=total)], capture_output=True, text=True, check=True, ) size, carried, codes, peak = completed.stdout.split() return int(size), int(carried), codes, int(peak) def test_a_declared_size_of_one_pixel_does_not_licence_an_unbounded_stream() -> None: """The review's repro, at the shipped bound: 1x1 declared, 400 MB paid.""" pytest.importorskip("pdfplumber") size, carried, codes, peak = _run_bomb(BOMB_STREAM_BYTES) assert size < 2 * 1024 * 1024, "the fixture must stay a small file, or it proves nothing" assert carried == 0, "a 400 MB stream was carried as a 1x1 picture" assert codes == "asset_too_large" assert peak < PEAK_RSS_BOUND, f"peak RSS {peak} bytes for a {size}-byte file" def test_the_refusal_reads_the_stream_and_not_only_the_declaration() -> None: """The mutation the shipped suite could not kill. An honest 20000x20000 declaration is refused by `check_size` alone, so a test built on one is green whether or not the actual stream is bounded. This document declares a size WITHIN the bound, which is the only shape that separates the two checks. """ pytest.importorskip("pdfplumber") small = zlib.compress(b"\x00" * 64, 9) within = extract_document("small.pdf", _bomb(8, payload=small), assets=True) assert [rejection.code for rejection in within.rejected] == [], "the control must be carried" assert len(within.images) == 1 def test_a_stream_over_the_bound_is_refused_with_a_patched_bound() -> None: """The same rule, cheap, so it runs on every machine and every suite.""" pytest.importorskip("pdfplumber") monkey = pytest.MonkeyPatch() try: monkey.setattr(assets, "MAX_IMAGE_BYTES", 4096) extracted = extract_document( "bomb.pdf", _bomb(1, payload=zlib.compress(b"\x00" * 1_000_000, 9)), assets=True ) finally: monkey.undo() assert extracted.images == () assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"] # --- MAJOR of the 18.09 review: a non-positive declaration is not a size ----- def test_a_non_positive_declared_size_is_refused_before_the_stream_is_read() -> None: """`-1 * 40_000_000_000` is NEGATIVE, so `pixels > MAX_IMAGE_PIXELS` was false and `check_size` returned silently; 400 MB was then decompressed and the refusal came from `encode_png` with `asset_samples_invalid` -- a code about the sample buffer for a defect in the declaration. The stream here is CORRUPT, so the order is observable: reading first gives `asset_pdf_unsupported`, reading the declaration first gives the new code. """ pytest.importorskip("pdfplumber") document = _bomb(4, payload=b"\xff" * 512).replace( b"/Width 4 /Height 4", b"/Width -1 /Height 40000000000" ) assert b"/Width -1" in document extracted = extract_document("negative.pdf", document, assets=True) assert extracted.images == () assert [rejection.code for rejection in extracted.rejected] == ["asset_size_invalid"] def test_check_size_refuses_every_non_positive_pair_and_keeps_unknown_unknown() -> None: for width, height in ((-1, 40_000_000_000), (0, 10), (10, 0), (-2, -2)): with pytest.raises(ExtractionError) as excinfo: assets.check_size(width, height, name="n") assert excinfo.value.code == "asset_size_invalid" # A size the container never declared is UNKNOWN, not invalid: there is no # number to bound and inventing one would refuse a legitimate picture. assets.check_size(None, None, name="n") assets.check_size(None, 10, name="n") # --- MINOR-3: the bound holds for a file carried verbatim, too --------------- def _png_header(width: int, height: int) -> bytes: """A PNG whose IHDR declares `width` x `height` and whose body is a stub. `read_image` sniffs and reads the header; it never decodes.""" def chunk(kind: bytes, payload: bytes) -> bytes: return ( len(payload).to_bytes(4, "big") + kind + payload + zlib.crc32(kind + payload).to_bytes(4, "big") ) ihdr = width.to_bytes(4, "big") + height.to_bytes(4, "big") + bytes([8, 0, 0, 0, 0]) return b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", ihdr) + chunk(b"IEND", b"") def test_an_image_file_over_the_bound_is_refused_although_it_is_never_decoded() -> None: """A 7 000 x 7 000 PNG is 49 MP in 47 705 bytes. This package does not decode a carried file, so it pays nothing -- but writing it into a bundle hands the consumer the same bomb with `7000x7000 px` printed beside it, and the README's first sentence says such an image is refused.""" with pytest.raises(ExtractionError) as excinfo: assets.read_image(_png_header(7000, 7000), name="big.png") assert excinfo.value.code == "asset_too_large" def test_an_image_file_under_the_bound_is_still_read() -> None: image = assets.read_image(_png_header(4515, 4128), name="drawing.png") assert (image.width, image.height) == (4515, 4128) # --- MINOR-1 and MINOR-2 of the 18.09 review -------------------------------- def test_a_remote_address_is_never_written_as_a_bare_url() -> None: """`inert` was only half true: the address was written TWICE, once in a code span and once bare, and a GFM/linkify renderer autolinks the bare one into ``. Measured with `markdown_it('gfm-like')`.""" line = render_missing(REMOTE, reason="the source is off this machine", href=REMOTE) assert "](" not in line assert REMOTE in line for position in range(len(line)): if line.startswith(REMOTE, position): assert line[position - 1] == "`" and line[position + len(REMOTE)] == "`", ( f"a bare occurrence of the address at {position}: {line!r}" ) def test_the_caption_of_a_remote_reference_is_still_stated() -> None: """`label` became a dead parameter in 0.10.1, so the alt text or figure caption of an image the bundle does not carry was DROPPED -- a regression against 0.10.0 and against this module's own reason for writing the line: a reader cannot weigh an absence they were never shown.""" line = render_missing( "p.gif", reason="the source is off this machine", label="Figur 84-1 Tverrprofil", href=None ) assert "Figur 84-1 Tverrprofil" in line with_href = render_missing( REMOTE, reason="the source is off this machine", label="Figur 84-1 Tverrprofil", href=REMOTE ) assert "Figur 84-1 Tverrprofil" in with_href