test(assets): the chain and the backstop, written red first
Eight tests for the two findings of the 18.09 PM checkpoint of `0f308c1`.
Seven are RED here and one is green-but-uncovered; the fix is the next commit.
BLOCKER -- `_check_inflated` read `filters[0]`, measured that one link and
returned, so a stream decoded through `/Filter [/FlateDecode /FlateDecode]`
was bounded by nothing. Reproduced here in its own interpreter, same commit:
links=1 408 516 B file -> 77 971 456 B peak RSS (bounded, 0.10.1)
links=2 1 636 B file -> 889 573 376 B peak RSS (unbounded)
links=3 1 070 B file -> 888 401 920 B peak RSS (unbounded)
543 000x the file size at two links, and the image IS refused at the end --
by `check_payload`, after the memory has been spent. PM measured the same
shape at 1 606 B -> 835 MB and 2 839 B -> 2 439 MB.
The chain is not a hypothetical. Over the 5 092 image objects of the 78 PDFs
on this machine (measured 2026-09-18): 1 625 `[/DCTDecode]`, 2 215
`[/FlateDecode]`, 596 `[/FlateDecode /DCTDecode]`, 580 `[/FlateDecode
/ASCII85Decode]`, 40 unfiltered, 16 `[/ASCII85Decode /FlateDecode]`, 16
`[/JPXDecode]`, 4 `[/CCITTFaxDecode]`. So refusing every chain would cost
1 192 real pictures, and bounding only the first link leaves those 16
unmeasured -- `filters[0]` is not `FlateDecode` there, so nothing ran at all.
Two of the cheap tests assert WHICH check fired ("decompresses to more than"),
because the backstop refuses the same document by code and a test reading only
the code is green on the defect.
MAJOR -- `check_payload(len(data), name=name)` after `get_data()` is the
counted refusal that four documentation surfaces point at, and deleting
exactly that line passed all 2 132 tests. The path that reaches it is a stream
pdfminer has ALREADY decoded (`decode()` sets `rawdata` to `None`), so there is
no raw stream left to measure. That test is green here and RED under the
deletion, measured before this commit: `asset_too_large` -> `asset_pdf_unsupported`.
Each red test carries a known-positive beside it: the two chains the corpora
hold still deliver their 64x64 picture, and an already-decoded stream under
the bound is still read.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
0f308c1f56
commit
3b587ea567
1 changed files with 279 additions and 9 deletions
|
|
@ -141,12 +141,34 @@ def _zeros_stream(total: int) -> bytes:
|
|||
return b"".join(parts)
|
||||
|
||||
|
||||
def _bomb(dimension: int, content: bytes | None = None, payload: bytes | None = None) -> bytes:
|
||||
def _nested_zeros_stream(total: int, links: int = 2) -> bytes:
|
||||
"""The same zeros, deflated `links` times over -- the shape a PDF declares
|
||||
as `/Filter [/FlateDecode /FlateDecode]`.
|
||||
|
||||
Deflate output of a long run of zeros is itself highly repetitive, so the
|
||||
second pass shrinks 407 685 bytes to 795: the whole 400 MB stands behind
|
||||
under a kilobyte of stream, which is why measuring only the first link of
|
||||
the chain is not a bound at all.
|
||||
"""
|
||||
stream = _zeros_stream(total)
|
||||
for _ in range(links - 1):
|
||||
stream = zlib.compress(stream, 9)
|
||||
return stream
|
||||
|
||||
|
||||
def _bomb(
|
||||
dimension: int,
|
||||
content: bytes | None = None,
|
||||
payload: bytes | None = None,
|
||||
filters: str = "/FlateDecode",
|
||||
) -> bytes:
|
||||
"""A tiny PDF declaring one `dimension` x `dimension` grayscale image of
|
||||
compressed zeros -- the review's own repro, built here. `content` replaces
|
||||
the page's content stream, for a page that draws an INLINE image instead.
|
||||
`payload` replaces the image stream, for a document whose DECLARED size and
|
||||
whose actual stream are two different numbers."""
|
||||
whose actual stream are two different numbers. `filters` replaces the
|
||||
`/Filter` entry, for a document whose stream is decoded by a CHAIN rather
|
||||
than by one filter."""
|
||||
if payload is None:
|
||||
payload = zlib.compress(b"\x00" * (dimension * dimension), 9)
|
||||
if content is None:
|
||||
|
|
@ -159,8 +181,8 @@ def _bomb(dimension: int, content: bytes | None = None, payload: bytes | None =
|
|||
b"<< /Length %d >>\nstream\n" % len(content) + content + b"\nendstream",
|
||||
(
|
||||
"<< /Type /XObject /Subtype /Image /Width %d /Height %d /ColorSpace /DeviceGray "
|
||||
"/BitsPerComponent 8 /Filter /FlateDecode /Length %d >>\nstream\n"
|
||||
% (dimension, dimension, len(payload))
|
||||
"/BitsPerComponent 8 /Filter %s /Length %d >>\nstream\n"
|
||||
% (dimension, dimension, filters, len(payload))
|
||||
).encode("ascii")
|
||||
+ payload
|
||||
+ b"\nendstream",
|
||||
|
|
@ -295,10 +317,10 @@ BOMB_STREAM_BYTES = 400 * 1024 * 1024
|
|||
_CHILD = """
|
||||
import resource, sys
|
||||
sys.path.insert(0, {tests!r})
|
||||
from test_asset_limits import _bomb, _zeros_stream
|
||||
from test_asset_limits import _bomb, _zeros_stream, _nested_zeros_stream
|
||||
from llm_ingestion_okf.extract import extract_document
|
||||
|
||||
document = _bomb(1, payload=_zeros_stream({total}))
|
||||
document = {builder}
|
||||
extracted = extract_document("bomb.pdf", document, assets=True)
|
||||
peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
|
||||
print(
|
||||
|
|
@ -310,11 +332,18 @@ print(
|
|||
"""
|
||||
|
||||
|
||||
def _run_bomb(total: int) -> tuple[int, int, str, int]:
|
||||
def _run_bomb(total: int, *, links: int = 1) -> tuple[int, int, str, int]:
|
||||
"""The bomb in its own interpreter, so peak RSS is ITS peak and not the
|
||||
high-water mark of every test that ran before it."""
|
||||
high-water mark of every test that ran before it. `links` is how many
|
||||
`FlateDecode` filters the stream is decoded by."""
|
||||
builder = (
|
||||
"_bomb(1, payload=_zeros_stream(%d))" % total
|
||||
if links == 1
|
||||
else "_bomb(1, payload=_nested_zeros_stream(%d, %d), filters=%r)"
|
||||
% (total, links, "[" + " ".join(["/FlateDecode"] * links) + "]")
|
||||
)
|
||||
completed = subprocess.run(
|
||||
[sys.executable, "-c", _CHILD.format(tests=str(Path(__file__).parent), total=total)],
|
||||
[sys.executable, "-c", _CHILD.format(tests=str(Path(__file__).parent), builder=builder)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=True,
|
||||
|
|
@ -460,3 +489,244 @@ def test_the_caption_of_a_remote_reference_is_still_stated() -> None:
|
|||
REMOTE, reason="the source is off this machine", label="Figur 84-1 Tverrprofil", href=REMOTE
|
||||
)
|
||||
assert "Figur 84-1 Tverrprofil" in with_href
|
||||
|
||||
|
||||
# --- BLOCKER of the 18.09 PM checkpoint: the CHAIN, not only its first link --
|
||||
#
|
||||
# `_check_inflated` read `filters[0]`, measured that one link, and returned.
|
||||
# A PDF may decode a stream through a chain (`/Filter [/FlateDecode
|
||||
# /FlateDecode]`), and the PM's repro measured what that costs on `0f308c1`:
|
||||
# 1 606 bytes of stream became 835 MB of peak RSS and 2 839 bytes became
|
||||
# 2 439 MB -- linear, about 860 000x the file size, and 400x worse than the
|
||||
# single-link bomb the 0.10.1 fix was written for. The image is refused at the
|
||||
# end by `check_payload`, after the memory has been spent; one 3 KB document
|
||||
# takes a batch build with it.
|
||||
#
|
||||
# The corpora say the chain is not a hypothetical shape. Over the 5 092 image
|
||||
# objects of the 78 PDFs on this machine (measured 2026-09-18), 596 decode
|
||||
# through `[/FlateDecode /DCTDecode]`, 580 through `[/FlateDecode
|
||||
# /ASCII85Decode]` and 16 through `[/ASCII85Decode /FlateDecode]` -- so a rule
|
||||
# that refuses every chain would cost 1 192 real pictures, and a rule that
|
||||
# bounds only the first link leaves all 16 of the last group unbounded.
|
||||
|
||||
|
||||
def test_a_chain_of_two_flate_filters_costs_no_more_than_one() -> None:
|
||||
"""The PM's repro at the shipped bound, in its own interpreter."""
|
||||
pytest.importorskip("pdfplumber")
|
||||
size, carried, codes, peak = _run_bomb(BOMB_STREAM_BYTES, links=2)
|
||||
assert size < 2 * 1024 * 1024, "the fixture must stay a small file, or it proves nothing"
|
||||
assert carried == 0, "a 400 MB stream was carried as a 1x1 picture"
|
||||
assert codes == "asset_too_large"
|
||||
assert peak < PEAK_RSS_BOUND, f"peak RSS {peak} bytes for a {size}-byte file"
|
||||
|
||||
|
||||
def test_a_chain_of_three_flate_filters_costs_no_more_than_one() -> None:
|
||||
"""A chain this package has never seen -- the bound must not be a list of
|
||||
the shapes somebody thought of."""
|
||||
pytest.importorskip("pdfplumber")
|
||||
size, carried, codes, peak = _run_bomb(BOMB_STREAM_BYTES, links=3)
|
||||
assert carried == 0
|
||||
assert codes == "asset_too_large"
|
||||
assert peak < PEAK_RSS_BOUND, f"peak RSS {peak} bytes for a {size}-byte file"
|
||||
|
||||
|
||||
def test_every_link_of_a_flate_chain_is_measured_with_a_patched_bound() -> None:
|
||||
"""The same rule, cheap, so it runs on every machine and every suite."""
|
||||
pytest.importorskip("pdfplumber")
|
||||
payload = _nested_zeros_stream(1 << 20, 2)
|
||||
monkey = pytest.MonkeyPatch()
|
||||
try:
|
||||
monkey.setattr(assets, "MAX_IMAGE_BYTES", 4096)
|
||||
extracted = extract_document(
|
||||
"nested.pdf",
|
||||
_bomb(1, payload=payload, filters="[/FlateDecode /FlateDecode]"),
|
||||
assets=True,
|
||||
)
|
||||
finally:
|
||||
monkey.undo()
|
||||
assert len(payload) < 4096, "the stream itself is under the bound; only its output is over"
|
||||
assert extracted.images == ()
|
||||
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
|
||||
# WHICH check fired is the whole point: the backstop refuses this document
|
||||
# too, after the memory is spent, so a test that reads only the code is
|
||||
# green on the defect. The two refusals say different things.
|
||||
assert "decompresses to more than" in extracted.rejected[0].reason, (
|
||||
"refused by the backstop after decoding, not by the bound before it: "
|
||||
+ extracted.rejected[0].reason
|
||||
)
|
||||
|
||||
|
||||
def test_a_flate_link_behind_an_ascii85_link_is_measured_too() -> None:
|
||||
"""`[/ASCII85Decode /FlateDecode]` is 16 real objects of the corpora, and
|
||||
`filters[0]` is not `FlateDecode`, so the old rule returned without
|
||||
measuring anything at all."""
|
||||
pytest.importorskip("pdfplumber")
|
||||
import base64
|
||||
|
||||
payload = base64.a85encode(_zeros_stream(1 << 20), adobe=True)
|
||||
monkey = pytest.MonkeyPatch()
|
||||
try:
|
||||
monkey.setattr(assets, "MAX_IMAGE_BYTES", 4096)
|
||||
extracted = extract_document(
|
||||
"a85.pdf",
|
||||
_bomb(1, payload=payload, filters="[/ASCII85Decode /FlateDecode]"),
|
||||
assets=True,
|
||||
)
|
||||
finally:
|
||||
monkey.undo()
|
||||
assert extracted.images == ()
|
||||
assert [rejection.code for rejection in extracted.rejected] == ["asset_too_large"]
|
||||
assert "decompresses to more than" in extracted.rejected[0].reason, (
|
||||
"refused by the backstop after decoding, not by the bound before it: "
|
||||
+ extracted.rejected[0].reason
|
||||
)
|
||||
|
||||
|
||||
def test_a_chained_image_within_the_bound_is_still_carried() -> None:
|
||||
"""The known-positive beside all of them: the two chains the corpora
|
||||
actually hold must still deliver their picture."""
|
||||
pytest.importorskip("pdfplumber")
|
||||
import base64
|
||||
|
||||
samples = zlib.compress(b"\x00" * (64 * 64), 9)
|
||||
for filters, payload in (
|
||||
("[/ASCII85Decode /FlateDecode]", base64.a85encode(samples, adobe=True)),
|
||||
("[/FlateDecode]", samples),
|
||||
):
|
||||
extracted = extract_document(
|
||||
"ok.pdf", _bomb(64, payload=payload, filters=filters), assets=True
|
||||
)
|
||||
assert [r.code for r in extracted.rejected] == [], filters
|
||||
assert [(i.width, i.height) for i in extracted.images] == [(64, 64)], filters
|
||||
|
||||
|
||||
def test_a_filter_the_bound_cannot_measure_is_refused_before_it_is_decoded() -> None:
|
||||
"""`LZWDecode` and `RunLengthDecode` expand by an amount pdfminer will only
|
||||
reveal by producing the whole output, so there is no measuring them a chunk
|
||||
at a time. They are refused with a code BEFORE the decode, which is the
|
||||
order the stream here makes observable: its bytes are not valid input for
|
||||
either filter, so decoding first gives `asset_pdf_unsupported`.
|
||||
"""
|
||||
pytest.importorskip("pdfplumber")
|
||||
for filters in ("/LZWDecode", "/RunLengthDecode", "[/FlateDecode /LZWDecode]"):
|
||||
extracted = extract_document(
|
||||
"unknown.pdf", _bomb(4, payload=b"\xff" * 512, filters=filters), assets=True
|
||||
)
|
||||
assert extracted.images == ()
|
||||
assert [r.code for r in extracted.rejected] == ["asset_pdf_unbounded"], filters
|
||||
|
||||
|
||||
def test_the_filters_this_package_bounds_are_the_ones_the_corpora_hold() -> None:
|
||||
"""The published claim, as a test: these names are what the close-out
|
||||
message and the four documentation surfaces say are carried."""
|
||||
from llm_ingestion_okf import extract as extract_module
|
||||
|
||||
carried = extract_module.bounded_pdf_filters()
|
||||
assert carried == {
|
||||
"FlateDecode",
|
||||
"ASCII85Decode",
|
||||
"ASCIIHexDecode",
|
||||
"DCTDecode",
|
||||
"JPXDecode",
|
||||
"JBIG2Decode",
|
||||
}
|
||||
|
||||
|
||||
def test_an_encrypted_stream_is_deciphered_and_then_bounded() -> None:
|
||||
"""`stream.decipher is not None` returned without measuring, so a document
|
||||
that declares encryption was a way past the bound. Deciphering does not
|
||||
change a stream's length, so this package does what pdfminer's own
|
||||
`decode()` does -- decipher first, then measure the filters.
|
||||
"""
|
||||
pytest.importorskip("pdfplumber")
|
||||
from pdfminer.pdftypes import PDFStream
|
||||
from pdfminer.psparser import LIT
|
||||
|
||||
from llm_ingestion_okf import extract as extract_module
|
||||
|
||||
def rot(objid: int, genno: int, data: bytes, attrs: dict[str, object]) -> bytes:
|
||||
return bytes(byte ^ 0x5A for byte in data)
|
||||
|
||||
raw = _zeros_stream(1 << 20)
|
||||
stream = PDFStream(
|
||||
{"Width": 1, "Height": 1, "BitsPerComponent": 8, "Filter": LIT("FlateDecode")},
|
||||
rot(0, 0, raw, {}),
|
||||
decipher=rot,
|
||||
)
|
||||
stream.set_objid(1, 0)
|
||||
monkey = pytest.MonkeyPatch()
|
||||
try:
|
||||
monkey.setattr(assets, "MAX_IMAGE_BYTES", 4096)
|
||||
with pytest.raises(ExtractionError) as excinfo:
|
||||
extract_module._check_stream_cost(stream, "encrypted")
|
||||
finally:
|
||||
monkey.undo()
|
||||
assert excinfo.value.code == "asset_too_large"
|
||||
|
||||
|
||||
# --- MAJOR of the 18.09 PM checkpoint: the backstop is covered ---------------
|
||||
|
||||
|
||||
def test_the_backstop_refuses_a_payload_the_stream_bound_never_saw() -> None:
|
||||
"""`check_payload(len(data))` after `get_data()` is the counted refusal the
|
||||
docstrings, the README, the CHANGELOG and CLAUDE.md all point at -- and
|
||||
deleting exactly that line passed the whole suite (2 132 tests) on
|
||||
`0f308c1`, because every other test reaches a bound that fires earlier.
|
||||
|
||||
The path that reaches it is a stream pdfminer has ALREADY decoded: `decode()`
|
||||
sets `rawdata` to `None`, so there is no raw stream left to measure and the
|
||||
memory is spent before this package is asked anything. The picture is then
|
||||
dropped by count rather than by bound, which is the weaker guarantee the
|
||||
documents describe -- and nothing held it.
|
||||
"""
|
||||
pytest.importorskip("pdfplumber")
|
||||
from pdfminer.pdftypes import PDFStream
|
||||
from pdfminer.psparser import LIT
|
||||
|
||||
from llm_ingestion_okf import extract as extract_module
|
||||
|
||||
stream = PDFStream(
|
||||
{"Width": 8, "Height": 8, "BitsPerComponent": 8, "Filter": LIT("FlateDecode")},
|
||||
zlib.compress(b"\x00" * 4096, 9),
|
||||
)
|
||||
stream.set_objid(1, 0)
|
||||
assert len(stream.get_data()) == 4096
|
||||
assert stream.get_rawdata() is None, (
|
||||
"the stream must be decoded already, or this proves nothing"
|
||||
)
|
||||
monkey = pytest.MonkeyPatch()
|
||||
try:
|
||||
monkey.setattr(assets, "MAX_IMAGE_BYTES", 1024)
|
||||
with pytest.raises(ExtractionError) as excinfo:
|
||||
extract_module._pdf_image(stream, "decoded")
|
||||
finally:
|
||||
monkey.undo()
|
||||
assert excinfo.value.code == "asset_too_large", (
|
||||
"without the backstop this falls through to the encoder and is reported "
|
||||
"as a defect of the sample buffer"
|
||||
)
|
||||
|
||||
|
||||
def test_a_decoded_stream_under_the_bound_is_still_read() -> None:
|
||||
"""The known-positive for the test above: the backstop must not be a rule
|
||||
that refuses every already-decoded stream."""
|
||||
pytest.importorskip("pdfplumber")
|
||||
from pdfminer.pdftypes import PDFStream
|
||||
from pdfminer.psparser import LIT
|
||||
|
||||
from llm_ingestion_okf import extract as extract_module
|
||||
|
||||
stream = PDFStream(
|
||||
{
|
||||
"Width": 8,
|
||||
"Height": 8,
|
||||
"BitsPerComponent": 8,
|
||||
"ColorSpace": LIT("DeviceGray"),
|
||||
"Filter": LIT("FlateDecode"),
|
||||
},
|
||||
zlib.compress(b"\x00" * 64, 9),
|
||||
)
|
||||
stream.set_objid(1, 0)
|
||||
stream.get_data()
|
||||
image = extract_module._pdf_image(stream, "decoded")
|
||||
assert (image.width, image.height) == (8, 8)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue