refactor(calibration): consolidate tunable thresholds into calibration.py
Session D: move every calibration constant (entropy floors 5.4/128, 5.1/64, 4.7/40 + shape floors; MAX_SCAN_CHARS; rot13-min; cognitive-load lengths 2000/2500; disposition ranks; active-content severities) into one documented calibration.py, so a parallel Node/TS port can mirror exactly the same numbers. Pure refactor, zero behavior change: calibration is a leaf module (imports only report.Severity) that entropy/lexicon/disposition/active_content now source their thresholds from. MAX_SCAN_CHARS is re-exported from lexicon so output.py and existing callers are unaffected. The 347 pre-existing tests pass unmodified; new test_calibration.py freezes the values and asserts each detector actually reads its threshold from calibration (identity-checked, not a dead copy).
This commit is contained in:
parent
4a9cfd2bbe
commit
ee402e4ea8
6 changed files with 212 additions and 34 deletions
99
tests/test_calibration.py
Normal file
99
tests/test_calibration.py
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
"""test_calibration — freeze the shared calibration surface (Session D).
|
||||
|
||||
Session D consolidated every tunable threshold into
|
||||
``llm_ingestion_guard.calibration`` so the Node port can mirror *exactly* the
|
||||
same numbers. These tests are the frozen contract in two halves:
|
||||
|
||||
1. the raw values themselves (the tuple the port shares), and
|
||||
2. the proof that each detector actually *sources* its threshold from here —
|
||||
so the freeze is a live single-source-of-truth, not a dead copy that can
|
||||
silently drift from the value the code uses.
|
||||
|
||||
Changing a calibration number is a deliberate recalibration: it must break a
|
||||
test here first.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from llm_ingestion_guard import calibration as cal
|
||||
from llm_ingestion_guard.report import Severity
|
||||
|
||||
|
||||
# --- frozen raw values ------------------------------------------------------
|
||||
|
||||
def test_entropy_thresholds_frozen():
|
||||
assert (cal.ENTROPY_CRITICAL_H, cal.ENTROPY_CRITICAL_LEN) == (5.4, 128)
|
||||
assert (cal.ENTROPY_HIGH_H, cal.ENTROPY_HIGH_LEN) == (5.1, 64)
|
||||
assert (cal.ENTROPY_MEDIUM_H, cal.ENTROPY_MEDIUM_LEN) == (4.7, 40)
|
||||
|
||||
|
||||
def test_entropy_shape_floors_frozen():
|
||||
assert cal.ENTROPY_BASE64_FLOOR_LEN == 100
|
||||
assert cal.ENTROPY_HEX_FLOOR_LEN == 64
|
||||
|
||||
|
||||
def test_lexicon_selfsafety_frozen():
|
||||
assert cal.MAX_SCAN_CHARS == 1_000_000
|
||||
assert cal.ROT13_MIN_LEN == 40
|
||||
|
||||
|
||||
def test_cognitive_load_lengths_frozen():
|
||||
assert cal.COGNITIVE_LOAD_MIN_LEN == 2500
|
||||
assert cal.COGNITIVE_LOAD_TAIL_START == 2000
|
||||
|
||||
|
||||
def test_disposition_rank_frozen():
|
||||
assert cal.DISPOSITION_RANK == {
|
||||
"warn": 0,
|
||||
"quarantine_review": 1,
|
||||
"fail_secure": 2,
|
||||
}
|
||||
|
||||
|
||||
def test_active_content_severity_frozen():
|
||||
assert cal.ACTIVE_CONTENT_SEVERITY == {
|
||||
"markdown-image": Severity.HIGH,
|
||||
"markdown-link": Severity.MEDIUM,
|
||||
"reference-link": Severity.MEDIUM,
|
||||
"autolink": Severity.MEDIUM,
|
||||
"raw-html": Severity.HIGH,
|
||||
"data-uri": Severity.HIGH,
|
||||
}
|
||||
|
||||
|
||||
# --- binding: each detector reads its threshold from calibration ------------
|
||||
# The freeze is meaningful only if the modules actually READ these values. An
|
||||
# import alias binds the SAME object, so identity (`is`) proves the single
|
||||
# source of truth rather than a coincidental equal copy.
|
||||
|
||||
def test_entropy_module_sources_from_calibration():
|
||||
from llm_ingestion_guard import entropy
|
||||
assert entropy._CRITICAL_H is cal.ENTROPY_CRITICAL_H
|
||||
assert entropy._CRITICAL_LEN is cal.ENTROPY_CRITICAL_LEN
|
||||
assert entropy._HIGH_H is cal.ENTROPY_HIGH_H
|
||||
assert entropy._HIGH_LEN is cal.ENTROPY_HIGH_LEN
|
||||
assert entropy._MEDIUM_H is cal.ENTROPY_MEDIUM_H
|
||||
assert entropy._MEDIUM_LEN is cal.ENTROPY_MEDIUM_LEN
|
||||
assert entropy._BASE64_FLOOR_LEN is cal.ENTROPY_BASE64_FLOOR_LEN
|
||||
assert entropy._HEX_FLOOR_LEN is cal.ENTROPY_HEX_FLOOR_LEN
|
||||
|
||||
|
||||
def test_lexicon_module_sources_from_calibration():
|
||||
from llm_ingestion_guard import lexicon
|
||||
assert lexicon.MAX_SCAN_CHARS is cal.MAX_SCAN_CHARS
|
||||
assert lexicon._ROT13_MIN_LEN is cal.ROT13_MIN_LEN
|
||||
|
||||
|
||||
def test_disposition_module_sources_from_calibration():
|
||||
from llm_ingestion_guard import disposition
|
||||
from llm_ingestion_guard.disposition import Disposition
|
||||
# Enum-keyed rank reconstructed from calibration's value-keyed source.
|
||||
assert disposition._DISPOSITION_RANK == {
|
||||
Disposition.WARN: 0,
|
||||
Disposition.QUARANTINE_REVIEW: 1,
|
||||
Disposition.FAIL_SECURE: 2,
|
||||
}
|
||||
|
||||
|
||||
def test_active_content_module_sources_from_calibration():
|
||||
from llm_ingestion_guard import active_content
|
||||
assert active_content._SEVERITY is cal.ACTIVE_CONTENT_SEVERITY
|
||||
Loading…
Add table
Add a link
Reference in a new issue