1 of 3 new guards RED, on an ASSERT about behaviour: `DID NOT RAISE ExtractionError` on all four cuts. The other two are the section's own known-positives and pass by describing what is already true -- where each cut lands in the fixture's opcode stream, and that the UNCOMPRESSED BMP path already refuses the same shape with `asset_samples_invalid`. Measured by PM 2026-09-19 on a real R761 asset (352x548 = 192 896 pixels): a stream cut to 90 % was carried with 13 923 pixels wrong, to 50 % with 95 890, to 10 % with 166 525 -- no code, no row, a partly blank PNG standing under a content address that says it holds the source's pixels. The test brings its OWN permissive RLE8 decoder, so the pixel cost of each cut is a number this file computes (8, 11, 0 and 0 of 32) rather than the package agreeing with itself; two of the four cuts lose no pixel at all, which is the whole reason the rule has to be the terminator and not a pixel count. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
614 lines
24 KiB
Python
614 lines
24 KiB
Python
"""Every image a bundle carries must be one a model can actually SHOW.
|
|
|
|
0.10.0 carried what a document declared, in whatever format the publisher
|
|
happened to ship. Measured 2026-09-19 over the frozen R761 delivery's own
|
|
`assets/` directory (denominator 50): 29 JPEG, 2 PNG and **19 "PC bitmap,
|
|
Windows 3.x, 8-bit, compression 1"** -- RLE8 BMP. The 19 are carried byte for
|
|
byte and are correct files; they are also files no model reads, so a figure
|
|
the bundle holds is a figure no arm can see. An absence a reader is shown is
|
|
information; a picture that is present and unreadable is worse than either,
|
|
because the count says it arrived.
|
|
|
|
THE PROPERTY, and it is a property rather than a list of formats we happen to
|
|
have met: for every asset in a built bundle, the format READ OFF ITS BYTES is
|
|
in the viewable set. A source image outside that set is converted losslessly
|
|
to PNG, or refused with a published code and a line in the concept saying what
|
|
stood there. Never carried silently in a format nothing can read.
|
|
|
|
The viewable set is stated here independently of the package, because a test
|
|
that imported the package's own constant would agree with it by construction
|
|
and measure nothing. Same for the rejection codes and the pixel grids below:
|
|
each guard is pinned against a count or a literal written out in this file, and
|
|
the lossless guard decodes through Pillow -- a decoder this package does not
|
|
use and does not ship -- so "identical pixels" is not this module agreeing with
|
|
itself.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import io
|
|
import struct
|
|
import warnings
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_okf import cli, corpus
|
|
from llm_ingestion_okf.assets import ASSETS_DIR, read_image, sniff
|
|
from llm_ingestion_okf.errors import ExtractionError
|
|
|
|
BUNDLE_ID = "asset-viewable-fixture"
|
|
OKF_VERSION = "0.2"
|
|
|
|
#: What a model can be shown. Written out here, never imported: the four
|
|
#: formats a vision-capable model decodes. A `.bmp`, a `.tiff` and a JPEG 2000
|
|
#: codestream are all legitimate image files and none of them is on this list.
|
|
VIEWABLE = frozenset({"image/png", "image/jpeg", "image/gif", "image/webp"})
|
|
|
|
#: The codes this file requires. Stated here so a rename of the package's own
|
|
#: constant cannot make these guards pass by agreeing with it.
|
|
CODE_NOT_VIEWABLE = "asset_not_viewable"
|
|
CODE_TOO_LARGE = "asset_too_large"
|
|
|
|
# --- fixtures built byte by byte, so the expected pixels are known ----------
|
|
|
|
#: Four RGB triples. Index 0 is white, which is also what an RLE delta skips
|
|
#: over, so the expected grid below states what a skip leaves behind.
|
|
PALETTE = bytes((255, 255, 255, 0, 0, 0, 200, 16, 32, 16, 96, 200))
|
|
|
|
#: An 8x4 image as palette indices, TOP-DOWN. Written out rather than
|
|
#: generated: this is the fasit the two decodings are both measured against.
|
|
EXPECTED_INDICES = (
|
|
(1, 1, 1, 1, 1, 1, 1, 1),
|
|
(0, 1, 2, 3, 3, 2, 1, 0),
|
|
(3, 3, 3, 0, 0, 1, 1, 1),
|
|
(0, 0, 0, 0, 0, 1, 2, 3),
|
|
)
|
|
|
|
#: The same image as an RLE8 opcode stream, BOTTOM-UP as the format requires,
|
|
#: exercising all five opcodes: encoded run, absolute run (with its pad byte),
|
|
#: delta, end-of-line and end-of-bitmap. A fixture using runs alone would pass
|
|
#: with a decoder that never implements the other three.
|
|
RLE8_STREAM = bytes(
|
|
(
|
|
0x05,
|
|
0x00,
|
|
0x00,
|
|
0x03,
|
|
0x01,
|
|
0x02,
|
|
0x03,
|
|
0x00,
|
|
0x00,
|
|
0x00,
|
|
0x03,
|
|
0x03,
|
|
0x00,
|
|
0x02,
|
|
0x02,
|
|
0x00,
|
|
0x03,
|
|
0x01,
|
|
0x00,
|
|
0x00,
|
|
0x00,
|
|
0x08,
|
|
0x00,
|
|
0x01,
|
|
0x02,
|
|
0x03,
|
|
0x03,
|
|
0x02,
|
|
0x01,
|
|
0x00,
|
|
0x00,
|
|
0x00,
|
|
0x08,
|
|
0x01,
|
|
0x00,
|
|
0x01,
|
|
)
|
|
)
|
|
|
|
|
|
def expected_rgb() -> bytes:
|
|
return b"".join(PALETTE[index * 3 : index * 3 + 3] for row in EXPECTED_INDICES for index in row)
|
|
|
|
|
|
def _dib(width: int, height: int, bits: int, compression: int, body: int, entries: int) -> bytes:
|
|
return struct.pack(
|
|
"<IiiHHIIiiII", 40, width, height, 1, bits, compression, body, 3779, 3779, entries, entries
|
|
)
|
|
|
|
|
|
def _file_header(offbits: int, body: int) -> bytes:
|
|
return b"BM" + struct.pack("<IHHI", offbits + body, 0, 0, offbits)
|
|
|
|
|
|
def bmp_rle8(width: int, height: int, palette: bytes, stream: bytes) -> bytes:
|
|
"""A Windows 3.x 8-bit RLE8 BMP, the shape 19 of R761's 50 assets have."""
|
|
entries = len(palette) // 3
|
|
table = b"".join(
|
|
bytes((palette[i * 3 + 2], palette[i * 3 + 1], palette[i * 3], 0)) for i in range(entries)
|
|
)
|
|
offbits = 14 + 40 + len(table)
|
|
return (
|
|
_file_header(offbits, len(stream))
|
|
+ _dib(width, height, 8, 1, len(stream), entries)
|
|
+ table
|
|
+ stream
|
|
)
|
|
|
|
|
|
def bmp_24(rows: tuple[tuple[tuple[int, int, int], ...], ...]) -> bytes:
|
|
"""An uncompressed 24-bit BMP -- the other half of the order's fixture."""
|
|
height = len(rows)
|
|
width = len(rows[0])
|
|
pad = b"\x00" * ((-width * 3) % 4)
|
|
body = b"".join(
|
|
b"".join(bytes((blue, green, red)) for (red, green, blue) in row) + pad
|
|
for row in reversed(rows)
|
|
)
|
|
offbits = 14 + 40
|
|
return _file_header(offbits, len(body)) + _dib(width, height, 24, 0, len(body), 0) + body
|
|
|
|
|
|
#: A minimal little-endian TIFF: one IFD with no usable entries. It is a real
|
|
#: TIFF by its magic and it is not a format this package can decode with the
|
|
#: stdlib, so it is the order's "cannot be converted" case.
|
|
TIFF_STUB = b"II\x2a\x00" + struct.pack("<I", 8) + struct.pack("<H", 0) + struct.pack("<I", 0)
|
|
|
|
RGB_ROWS = (
|
|
((255, 0, 0), (0, 255, 0), (0, 0, 255), (255, 255, 0), (0, 0, 0)),
|
|
((1, 2, 3), (4, 5, 6), (7, 8, 9), (10, 11, 12), (13, 14, 15)),
|
|
((255, 255, 255), (128, 128, 128), (0, 0, 0), (17, 34, 51), (68, 85, 102)),
|
|
)
|
|
|
|
|
|
def png_1x1() -> bytes:
|
|
"""A real PNG, built with zlib alone -- the already-viewable control."""
|
|
import zlib
|
|
|
|
def chunk(kind: bytes, payload: bytes) -> bytes:
|
|
return (
|
|
len(payload).to_bytes(4, "big")
|
|
+ kind
|
|
+ payload
|
|
+ zlib.crc32(kind + payload).to_bytes(4, "big")
|
|
)
|
|
|
|
ihdr = struct.pack(">IIBBBBB", 1, 1, 8, 2, 0, 0, 0)
|
|
return (
|
|
b"\x89PNG\r\n\x1a\n"
|
|
+ chunk(b"IHDR", ihdr)
|
|
+ chunk(b"IDAT", zlib.compress(b"\x00\x10\x20\x30", 9))
|
|
+ chunk(b"IEND", b"")
|
|
)
|
|
|
|
|
|
def jpeg_bytes() -> bytes:
|
|
"""A real baseline JPEG, encoded by Pillow -- the second viewable control."""
|
|
Image = pytest.importorskip("PIL.Image")
|
|
buffer = io.BytesIO()
|
|
Image.new("RGB", (4, 3), (10, 20, 30)).save(buffer, format="JPEG")
|
|
return buffer.getvalue()
|
|
|
|
|
|
def _try_decode(data: bytes) -> tuple[tuple[int, int], bytes] | None:
|
|
"""`decode_rgb`, or `None` where the independent decoder cannot read it.
|
|
|
|
Needed because this guard scans EVERY carried asset, and the point of the
|
|
round is that a bundle may hold one nothing can decode. A raised
|
|
`UnidentifiedImageError` would make the guard red for the helper's reason
|
|
instead of for the claim's.
|
|
"""
|
|
try:
|
|
return decode_rgb(data)
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def decode_rgb(data: bytes) -> tuple[tuple[int, int], bytes]:
|
|
"""An INDEPENDENT decoding: Pillow, which this package neither uses nor ships."""
|
|
Image = pytest.importorskip("PIL.Image")
|
|
with Image.open(io.BytesIO(data)) as image:
|
|
return image.size, image.convert("RGB").tobytes()
|
|
|
|
|
|
# --- the inbox and the build ------------------------------------------------
|
|
|
|
|
|
def _inbox(root: Path) -> Path:
|
|
inbox = root / "inbox"
|
|
(inbox / "graphics").mkdir(parents=True)
|
|
(inbox / "graphics" / "figur-rle8.bmp").write_bytes(bmp_rle8(8, 4, PALETTE, RLE8_STREAM))
|
|
(inbox / "graphics" / "figur-flat.bmp").write_bytes(bmp_24(RGB_ROWS))
|
|
(inbox / "graphics" / "skjema.tiff").write_bytes(TIFF_STUB)
|
|
(inbox / "prosess.html").write_text(
|
|
"<!doctype html>\n<html><head><title>Prosess 84</title></head>\n<body>\n"
|
|
"<h1>84 Konstruksjoner av betong</h1>\n"
|
|
"<p>Toleranseklassene staar i figuren under.</p>\n"
|
|
'<img src="graphics/figur-rle8.bmp" alt="Figur 84-1 Toleranseklasser">\n'
|
|
"<p>Og prinsippet:</p>\n"
|
|
'<img src="graphics/figur-flat.bmp" alt="Figur 84-2 Prinsipp">\n'
|
|
"<p>Og skjemaet:</p>\n"
|
|
'<img src="graphics/skjema.tiff" alt="Skjema 84-3">\n'
|
|
"</body></html>\n",
|
|
encoding="utf-8",
|
|
)
|
|
return inbox
|
|
|
|
|
|
def _build(inbox: Path, bundle: Path, *extra: str) -> int:
|
|
with warnings.catch_warnings():
|
|
warnings.simplefilter("ignore")
|
|
return cli.main(
|
|
[
|
|
"build",
|
|
str(inbox),
|
|
"--bundle",
|
|
str(bundle),
|
|
"--bundle-id",
|
|
BUNDLE_ID,
|
|
"--okf-version",
|
|
OKF_VERSION,
|
|
*extra,
|
|
]
|
|
)
|
|
|
|
|
|
def _concepts(bundle: Path) -> list[Path]:
|
|
return [
|
|
path
|
|
for path in sorted(bundle.rglob("*.md"))
|
|
if path.name not in {"index.md", corpus.LOG_NAME}
|
|
]
|
|
|
|
|
|
def _assets(bundle: Path) -> list[Path]:
|
|
directory = bundle / ASSETS_DIR
|
|
return sorted(p for p in directory.iterdir() if p.is_file()) if directory.is_dir() else []
|
|
|
|
|
|
# --- 1. visibility, as a property of every carried asset --------------------
|
|
|
|
|
|
def test_every_carried_asset_is_a_format_a_model_can_show(tmp_path: Path) -> None:
|
|
"""Eval point 1. RED before the conversion: two BMPs are carried as BMP.
|
|
|
|
The denominator is counted in this test from the inbox itself, so a build
|
|
that silently carried nothing would fail here rather than pass over an
|
|
empty set -- the failure mode a "0 of 0 violations" assertion cannot see.
|
|
"""
|
|
pytest.importorskip("llm_ingestion_guard")
|
|
inbox = _inbox(tmp_path)
|
|
dropped = [p for p in sorted((inbox / "graphics").iterdir()) if p.is_file()]
|
|
assert len(dropped) == 3, dropped
|
|
not_viewable_in = [p for p in dropped if (sniff(p.read_bytes()) or ("", ""))[0] not in VIEWABLE]
|
|
assert len(not_viewable_in) == 3, not_viewable_in
|
|
|
|
bundle = tmp_path / "bundle"
|
|
assert _build(inbox, bundle) == 0
|
|
carried = _assets(bundle)
|
|
assert carried, "the build carried no asset at all"
|
|
offending = [
|
|
(path.name, (sniff(path.read_bytes()) or (None,))[0])
|
|
for path in carried
|
|
if (sniff(path.read_bytes()) or (None,))[0] not in VIEWABLE
|
|
]
|
|
assert offending == [], (
|
|
f"{len(offending)} of {len(carried)} carried assets are in a format no model "
|
|
f"can show: {offending}"
|
|
)
|
|
|
|
|
|
# --- 2. lossless, against a decoder this package does not use ---------------
|
|
|
|
|
|
def test_the_rle8_source_decodes_to_the_grid_this_file_states() -> None:
|
|
"""The fixture's own known-positive: without it, guard 2 could compare two
|
|
wrong decodings and agree."""
|
|
size, rgb = decode_rgb(bmp_rle8(8, 4, PALETTE, RLE8_STREAM))
|
|
assert size == (8, 4)
|
|
assert rgb == expected_rgb()
|
|
|
|
|
|
def test_the_carried_png_holds_the_source_pixels_exactly(tmp_path: Path) -> None:
|
|
"""Eval point 2. Same RGB per pixel, through the palette, 8x4 = 32 pixels."""
|
|
pytest.importorskip("llm_ingestion_guard")
|
|
bundle = tmp_path / "bundle"
|
|
assert _build(_inbox(tmp_path), bundle) == 0
|
|
sources = {
|
|
"rle8": bmp_rle8(8, 4, PALETTE, RLE8_STREAM),
|
|
"flat24": bmp_24(RGB_ROWS),
|
|
}
|
|
carried = {path.name: path.read_bytes() for path in _assets(bundle)}
|
|
assert len(carried) >= 2, carried
|
|
|
|
matched = 0
|
|
for label, source in sources.items():
|
|
source_size, source_rgb = decode_rgb(source)
|
|
# VIEWABLE is part of the claim, not decoration: a BMP carried verbatim
|
|
# trivially holds its own pixels, so a guard that did not require the
|
|
# holder to be viewable would be green before the conversion exists.
|
|
hits = [
|
|
data
|
|
for data in carried.values()
|
|
if (sniff(data) or (None,))[0] in VIEWABLE
|
|
and _try_decode(data) == (source_size, source_rgb)
|
|
]
|
|
assert len(hits) == 1, f"{label}: {len(hits)} carried assets hold these pixels"
|
|
matched += 1
|
|
assert matched == 2
|
|
assert len(expected_rgb()) == 8 * 4 * 3
|
|
|
|
|
|
# --- 3. the bundle says what happened ---------------------------------------
|
|
|
|
|
|
def test_the_bundle_records_the_conversion_per_asset(tmp_path: Path) -> None:
|
|
"""Eval point 3: original format, original checksum, new checksum."""
|
|
pytest.importorskip("llm_ingestion_guard")
|
|
bundle = tmp_path / "bundle"
|
|
assert _build(_inbox(tmp_path), bundle) == 0
|
|
text = "\n".join(path.read_text("utf-8") for path in _concepts(bundle))
|
|
sources = (bmp_rle8(8, 4, PALETTE, RLE8_STREAM), bmp_24(RGB_ROWS))
|
|
for source in sources:
|
|
before = hashlib.sha256(source).hexdigest()
|
|
assert before in text, f"the bundle never states the source checksum {before[:12]}"
|
|
assert text.count("image/bmp") >= 2, "the bundle never states the original format"
|
|
for path in _assets(bundle):
|
|
after = hashlib.sha256(path.read_bytes()).hexdigest()
|
|
assert path.name.startswith(after[:12]), path.name
|
|
converted = [p for p in _assets(bundle) if p.suffix == ".png"]
|
|
assert len(converted) == 2, converted
|
|
for path in converted:
|
|
assert hashlib.sha256(path.read_bytes()).hexdigest() in text
|
|
|
|
|
|
def test_a_converted_asset_is_one_asset_and_not_two(tmp_path: Path) -> None:
|
|
"""The accounting's rule: converting is not a second element.
|
|
|
|
Two BMPs in, two files in `assets/`, two pointers, and the concept's own
|
|
`images:` count agreeing with the pointers it holds.
|
|
"""
|
|
pytest.importorskip("llm_ingestion_guard")
|
|
bundle = tmp_path / "bundle"
|
|
assert _build(_inbox(tmp_path), bundle) == 0
|
|
assert len(_assets(bundle)) == 2, _assets(bundle)
|
|
pointers = sum(path.read_text("utf-8").count("](/assets/") for path in _concepts(bundle))
|
|
assert pointers == 2, pointers
|
|
for path in _concepts(bundle):
|
|
body = path.read_text("utf-8")
|
|
if "\nimages: " in body:
|
|
line = next(row for row in body.splitlines() if row.startswith("images: "))
|
|
assert int(line.split(":", 1)[1]) == body.count("](/assets/")
|
|
|
|
|
|
# --- 4. the ceiling is checked before the pixels exist ----------------------
|
|
|
|
|
|
def test_a_bmp_declaring_a_huge_size_is_refused_before_it_is_decoded() -> None:
|
|
"""Eval point 4. The bound is on the DECLARATION, paid before the pixels."""
|
|
import tracemalloc
|
|
|
|
# 49 MP, just over the package's 40 MP bound. Deliberately NOT 50 000 x
|
|
# 50 000: the guard has to stay able to tell "refused before the decode"
|
|
# from "refused after it", and measuring the second costs a frame of the
|
|
# declared size. At 7 000 x 7 000 that frame is 49 MB -- affordable to
|
|
# measure, and two orders of magnitude over the 4 MB this asserts.
|
|
bomb = bmp_rle8(7_000, 7_000, PALETTE, bytes([0xFF, 0x01] * 64))
|
|
assert 7_000 * 7_000 > 40_000_000
|
|
assert len(bomb) < 4096, len(bomb)
|
|
tracemalloc.start()
|
|
try:
|
|
with pytest.raises(ExtractionError) as caught:
|
|
read_image(bomb, name="bombe.bmp")
|
|
finally:
|
|
_, peak = tracemalloc.get_traced_memory()
|
|
tracemalloc.stop()
|
|
assert caught.value.code == CODE_TOO_LARGE, caught.value.code
|
|
assert peak < 4 * 1024 * 1024, peak
|
|
|
|
|
|
def test_an_rle_stream_longer_than_its_declared_image_costs_the_declaration() -> None:
|
|
"""An RLE8 run list that paints far past an 8x4 frame pays for 8x4."""
|
|
import tracemalloc
|
|
|
|
flood = bytes([0xFF, 0x02] * 400_000) + bytes((0x00, 0x01))
|
|
bomb = bmp_rle8(8, 4, PALETTE, flood)
|
|
assert len(bomb) > 800_000, len(bomb)
|
|
tracemalloc.start()
|
|
try:
|
|
try:
|
|
image = read_image(bomb, name="flom.bmp")
|
|
except ExtractionError as exc:
|
|
code = exc.code
|
|
image = None
|
|
else:
|
|
code = None
|
|
finally:
|
|
_, peak = tracemalloc.get_traced_memory()
|
|
tracemalloc.stop()
|
|
if image is not None:
|
|
assert sniff(image.data) is not None
|
|
assert (sniff(image.data) or (None,))[0] in VIEWABLE
|
|
else:
|
|
assert code is not None
|
|
# The whole file is 800 KB; decoding it must not cost a multiple of that.
|
|
assert peak < 16 * 1024 * 1024, peak
|
|
|
|
|
|
def test_a_bmp_whose_name_claims_png_is_still_converted() -> None:
|
|
"""The type is read off the BYTES. A rule that trusted the extension would
|
|
carry this file unchanged and call it viewable, and every structural check
|
|
downstream would agree with it -- the count, the suffix and the pointer.
|
|
|
|
The R761 delivery is why this is not hypothetical: its graphics directory
|
|
holds `.bmp`, `.jpg` and `.png` side by side and the document's own
|
|
`xlink:href` values are whatever the publisher's tool wrote.
|
|
"""
|
|
image = read_image(bmp_rle8(8, 4, PALETTE, RLE8_STREAM), name="graphics/figur.png")
|
|
assert image.media_type == "image/png"
|
|
assert image.suffix == ".png"
|
|
assert decode_rgb(image.data) == ((8, 4), expected_rgb())
|
|
|
|
|
|
# --- 5. the known-positive: what is already viewable does not move ----------
|
|
|
|
|
|
def test_a_png_is_carried_byte_for_byte() -> None:
|
|
"""Eval point 5, in-test half. Conversion must reach nothing already viewable."""
|
|
data = png_1x1()
|
|
image = read_image(data, name="figur.png")
|
|
assert image.data == data
|
|
assert image.media_type == "image/png"
|
|
assert image.suffix == ".png"
|
|
|
|
|
|
def test_a_jpeg_is_carried_byte_for_byte() -> None:
|
|
data = jpeg_bytes()
|
|
image = read_image(data, name="figur.jpg")
|
|
assert image.data == data
|
|
assert image.media_type == "image/jpeg"
|
|
assert image.suffix == ".jpg"
|
|
|
|
|
|
# --- 6. what cannot be converted is refused out loud ------------------------
|
|
|
|
|
|
def test_a_tiff_is_refused_with_a_published_code() -> None:
|
|
"""Eval point 6: never silently carried in a format nothing reads."""
|
|
with pytest.raises(ExtractionError) as caught:
|
|
read_image(TIFF_STUB, name="skjema.tiff")
|
|
assert caught.value.code == CODE_NOT_VIEWABLE, caught.value.code
|
|
|
|
|
|
def test_a_corrupt_bmp_is_refused_with_a_published_code() -> None:
|
|
"""A BMP header with a body that stops mid-stream is a refusal, not a guess."""
|
|
truncated = bmp_rle8(8, 4, PALETTE, RLE8_STREAM)[:60]
|
|
with pytest.raises(ExtractionError) as caught:
|
|
read_image(truncated, name="kuttet.bmp")
|
|
assert caught.value.code.startswith("asset_"), caught.value.code
|
|
|
|
|
|
def test_the_refused_image_leaves_a_row_the_reader_can_see(tmp_path: Path) -> None:
|
|
"""A refusal is a line in the concept and a row in the run log."""
|
|
pytest.importorskip("llm_ingestion_guard")
|
|
bundle = tmp_path / "bundle"
|
|
assert _build(_inbox(tmp_path), bundle) == 0
|
|
text = "\n".join(path.read_text("utf-8") for path in _concepts(bundle))
|
|
assert "not carried" in text
|
|
assert "skjema.tiff" in text
|
|
log = (bundle / corpus.LOG_NAME).read_text("utf-8")
|
|
assert "image" in log.lower()
|
|
|
|
|
|
# --- 7. a truncated RLE8 stream is refused, never painted -------------------
|
|
|
|
#: The code the UNCOMPRESSED path already raises for a body that stops short
|
|
#: ("refusing to pad, because a short buffer means the header was read wrong").
|
|
#: Written out here rather than imported, like the two codes above.
|
|
CODE_SAMPLES_INVALID = "asset_samples_invalid"
|
|
|
|
|
|
def rle8_indices(stream: bytes, width: int, height: int) -> bytes:
|
|
"""A PERMISSIVE RLE8 decoder, written in this file, top-down.
|
|
|
|
It paints what the stream holds and leaves the rest at index 0. That is
|
|
what the format says about a SKIPPED pixel and what every decoder does
|
|
with a stream that simply stops -- which is exactly why a truncation is
|
|
invisible without a check: no decoder disagrees, they all agree on the
|
|
wrong picture. Its job here is to MEASURE what each cut costs, so the
|
|
guard below is pinned to a pixel count this file computed rather than to
|
|
the package's opinion of itself.
|
|
"""
|
|
rows = [bytearray(width) for _ in range(height)]
|
|
position, end, x, y = 0, len(stream), 0, 0
|
|
while position + 1 < end:
|
|
count, value = stream[position], stream[position + 1]
|
|
position += 2
|
|
if count:
|
|
if 0 <= y < height and x < width:
|
|
stop = min(x + count, width)
|
|
rows[y][x:stop] = bytes((value,)) * (stop - x)
|
|
x += count
|
|
continue
|
|
if value == 0:
|
|
x, y = 0, y + 1
|
|
elif value == 1:
|
|
break
|
|
elif value == 2:
|
|
if position + 2 > end:
|
|
break
|
|
x += stream[position]
|
|
y += stream[position + 1]
|
|
position += 2
|
|
else:
|
|
run = stream[position : position + value]
|
|
position += value + (value & 1)
|
|
if 0 <= y < height and x < width:
|
|
stop = min(x + len(run), width)
|
|
rows[y][x:stop] = run[: stop - x]
|
|
x += value
|
|
rows.reverse()
|
|
return b"".join(bytes(row) for row in rows)
|
|
|
|
|
|
def test_the_cuts_below_land_on_the_opcodes_this_file_names() -> None:
|
|
"""The fixture's own known-positive: a cut that no longer truncates
|
|
anything would make every guard below pass over nothing."""
|
|
assert len(RLE8_STREAM) == 36
|
|
assert RLE8_STREAM[32:34] == bytes((0x08, 0x01)), "the last encoded run"
|
|
assert RLE8_STREAM[20:22] == bytes((0x00, 0x08)), "the absolute block"
|
|
assert RLE8_STREAM[34:36] == bytes((0x00, 0x01)), "end-of-bitmap"
|
|
assert rle8_indices(RLE8_STREAM, 8, 4) == bytes(
|
|
index for row in EXPECTED_INDICES for index in row
|
|
)
|
|
|
|
|
|
def test_a_truncated_rle8_stream_is_refused_with_a_published_code() -> None:
|
|
"""Eval point 6 again, on the one case that was silent.
|
|
|
|
Measured by PM 2026-09-19 on a real R761 asset (352x548 = 192 896 pixels):
|
|
a stream cut to 90 % carried with 13 923 pixels wrong, to 50 % with 95 890,
|
|
to 10 % with 166 525 -- no code, no row, a partly blank PNG under a name
|
|
that says it holds the source's pixels. The uncompressed path already
|
|
refuses the same shape. NO PIXEL MAY BE GUESSED: either every one of them
|
|
is decoded from the stream, or the picture is refused.
|
|
|
|
THE RULE IS THE TERMINATOR, and it is read off the corpus rather than
|
|
chosen: over the 19 real RLE8 assets of the frozen R761 delivery, 19 of 19
|
|
end at an explicit end-of-bitmap escape, that escape is the stream's LAST
|
|
two bytes on 19 of 19, and `biSizeImage` equals the available bytes on
|
|
19 of 19. So a stream that runs out before its terminator is refused, and
|
|
so is one that is whole but never states it is finished -- the two are the
|
|
same bytes from a reader's side, and the measurement says no real writer
|
|
here produces the second.
|
|
"""
|
|
full = rle8_indices(RLE8_STREAM, 8, 4)
|
|
cuts = {
|
|
"cut in the middle of an encoded run": (RLE8_STREAM[:33], 8),
|
|
"cut in the middle of an absolute block": (RLE8_STREAM[:26], 11),
|
|
"cut immediately before end-of-bitmap": (RLE8_STREAM[:34], 0),
|
|
"ends with end-of-line instead of end-of-bitmap": (RLE8_STREAM[:34] + bytes((0, 0)), 0),
|
|
}
|
|
for label, (stream, lost) in cuts.items():
|
|
differ = sum(a != b for a, b in zip(full, rle8_indices(stream, 8, 4)))
|
|
assert differ == lost, f"{label}: {differ} pixels differ, this file says {lost}"
|
|
with pytest.raises(ExtractionError) as caught:
|
|
read_image(bmp_rle8(8, 4, PALETTE, stream), name="kuttet.bmp")
|
|
assert caught.value.code == CODE_SAMPLES_INVALID, f"{label}: {caught.value.code}"
|
|
|
|
# The control: the refusal is about the cut, not about the fixture.
|
|
whole = read_image(bmp_rle8(8, 4, PALETTE, RLE8_STREAM), name="hel.bmp")
|
|
assert whole.media_type == "image/png"
|
|
|
|
|
|
def test_the_uncompressed_path_refuses_the_same_shape() -> None:
|
|
"""The twin this rule was aligned to, so the two BMP paths cannot drift."""
|
|
short = bmp_24(RGB_ROWS)[:-30]
|
|
with pytest.raises(ExtractionError) as caught:
|
|
read_image(short, name="kort.bmp")
|
|
assert caught.value.code == CODE_SAMPLES_INVALID, caught.value.code
|