The context sets, the packaged knowledge bases and the example bundles are replaced by one fictitious example set about IT operations in an invented organisation: three context sets (serverrom-2027, driftsavtale-2027 and the two-base drift-og-avtale-2027), two synthetic knowledge bases under src/portfolio_optimiser/data/kunnskapsbaser and two example bundles under src/portfolio_optimiser/data/bundles. Numbers, codes and structural values in tests and fixtures are kept; names, ids and wording change. Dated measurement documents that only recorded runs on the replaced material are deleted. Gate figures measured on the new set are not comparable with earlier ones. The exclusion gate from the previous commit is green: 0 tracked files hit outside the shared/ subtree. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
274 lines
11 KiB
Python
274 lines
11 KiB
Python
"""Step 3 tests — non-tautological top-K retrieval + REAL-SessionContext two-arg injection.
|
|
|
|
The true match shares the *structured* similarity fields with the query but uses different
|
|
description text; the decoys share surface text but differ structurally. The injection test
|
|
uses a REAL ``agent_framework.SessionContext`` (not a single-arg fake), exercising the
|
|
genuine two-arg ``extend_instructions(source_id, instructions)`` GA signature — retiring the
|
|
Critical Fase 1 risk. Pattern: tests/spikes/test_d_verdictstore.py + real SessionContext.
|
|
"""
|
|
|
|
import json
|
|
import logging
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
from agent_framework import SessionContext
|
|
|
|
from portfolio_optimiser.verdicts import (
|
|
ExpeLContextProvider,
|
|
ProposalFeatures,
|
|
Verdict,
|
|
VerdictStore,
|
|
bundle_candidate_features,
|
|
capture_verdict,
|
|
load_verdicts_from_dir,
|
|
seed_store_from_bundle,
|
|
write_verdict,
|
|
)
|
|
|
|
_BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
|
|
|
|
_QUERY = ProposalFeatures(
|
|
affected_codes=frozenset({"05.2", "03.1"}),
|
|
measure_type="scope_reduction",
|
|
claimed_saving_nok=220_000, # bucket [100k, 500k)
|
|
description="licence scope reduction near head office",
|
|
)
|
|
|
|
|
|
def _store_with_true_match_and_decoys() -> tuple[VerdictStore, str]:
|
|
true_match = Verdict(
|
|
id="TRUE",
|
|
proposal_features=ProposalFeatures(
|
|
affected_codes=frozenset({"05.2", "03.1"}), # same codes
|
|
measure_type="scope_reduction", # same measure type
|
|
claimed_saving_nok=200_000, # same magnitude bucket
|
|
description="zzz totally unrelated wording alpha beta", # DIFFERENT text
|
|
),
|
|
decision="approved",
|
|
rationale="prior scope reduction on the same codes was approved",
|
|
)
|
|
decoy_low = Verdict(
|
|
id="DECOY-LOW",
|
|
proposal_features=ProposalFeatures(
|
|
affected_codes=frozenset({"09.1"}),
|
|
measure_type="rate_renegotiation",
|
|
claimed_saving_nok=50_000,
|
|
description="licence scope reduction near head office", # same words as query
|
|
),
|
|
decision="rejected",
|
|
rationale="surface-text decoy",
|
|
)
|
|
decoy_high = Verdict(
|
|
id="DECOY-HIGH",
|
|
proposal_features=ProposalFeatures(
|
|
affected_codes=frozenset({"21.2"}),
|
|
measure_type="material_substitution",
|
|
claimed_saving_nok=700_000,
|
|
description="licence scope reduction extra words",
|
|
),
|
|
decision="rejected",
|
|
rationale="surface-text decoy",
|
|
)
|
|
return VerdictStore(verdicts=[decoy_low, true_match, decoy_high]), "TRUE"
|
|
|
|
|
|
def test_retrieve_finds_structural_match_over_text_decoys() -> None:
|
|
store, true_id = _store_with_true_match_and_decoys()
|
|
hits = store.retrieve(_QUERY, k=3)
|
|
assert hits[0].id == true_id # structural match ranks #1 despite different wording
|
|
|
|
|
|
def test_retrieve_is_deterministic() -> None:
|
|
store, _ = _store_with_true_match_and_decoys()
|
|
assert [h.id for h in store.retrieve(_QUERY, k=3)] == [
|
|
h.id for h in store.retrieve(_QUERY, k=3)
|
|
]
|
|
|
|
|
|
def test_retrieve_rejects_non_positive_k() -> None:
|
|
store, _ = _store_with_true_match_and_decoys()
|
|
with pytest.raises(ValueError):
|
|
store.retrieve(_QUERY, k=0)
|
|
|
|
|
|
def test_capture_verdict_mints_stable_id() -> None:
|
|
a = capture_verdict(_QUERY, "approved", "ok")
|
|
b = capture_verdict(
|
|
ProposalFeatures(
|
|
affected_codes=frozenset({"03.1", "05.2"}), # same set, different order
|
|
measure_type="scope_reduction",
|
|
claimed_saving_nok=220_000,
|
|
description="DIFFERENT surface wording entirely", # text excluded from the id
|
|
),
|
|
"approved",
|
|
"ok",
|
|
)
|
|
assert a.id == b.id # structurally identical -> stable id
|
|
assert len(a.id) == 16
|
|
|
|
|
|
async def test_before_run_populates_real_sessioncontext_two_arg() -> None:
|
|
store, true_id = _store_with_true_match_and_decoys()
|
|
provider = ExpeLContextProvider(store, _QUERY, k=2)
|
|
# A REAL SessionContext (not a single-arg fake) — exercises the genuine GA two-arg
|
|
# extend_instructions(source_id, instructions) signature.
|
|
ctx = SessionContext(input_messages=[], instructions=[])
|
|
await provider.before_run(agent=None, session=None, context=ctx, state={})
|
|
assert any(true_id in instr for instr in ctx.instructions)
|
|
|
|
|
|
# --- OKF-bundle seeding (Fase 2a): the pre-hypothesis ExpeL query key + the seed store ---
|
|
|
|
|
|
def test_bundle_candidate_features_keys_on_the_ir_projection() -> None:
|
|
"""The pre-hypothesis ExpeL query is the candidate measure's cost-IR features (from the
|
|
bundle's ``validator-input.json``) — available BEFORE any proposal is generated."""
|
|
features = bundle_candidate_features(str(_BUNDLE_DIR))
|
|
assert features.affected_codes == frozenset({"ENERGI-TOTAL-EL"})
|
|
assert "LED-retrofit" in features.measure_type
|
|
assert features.claimed_saving_nok == 30000
|
|
|
|
|
|
def test_seed_store_from_bundle_carries_the_realization_signal() -> None:
|
|
"""Each ``type: verdict`` file becomes a structurally-keyed ``Verdict`` whose rationale carries
|
|
the learning signal the validator cannot compute (the realization rate 0.82)."""
|
|
store = seed_store_from_bundle(str(_BUNDLE_DIR))
|
|
assert len(store.verdicts) == 1
|
|
seed = store.verdicts[0]
|
|
assert seed.proposal_features.affected_codes == frozenset({"ENERGI-TOTAL-EL"})
|
|
assert "approved" in seed.decision
|
|
assert "0.82" in seed.rationale
|
|
|
|
|
|
def test_seed_store_retrieval_matches_the_candidate() -> None:
|
|
"""A3: seed and query derive from the SAME IR -> similarity 1.0 -> the lone seed is retrieved
|
|
for the candidate (the structural match the Step-1 wiring relies on)."""
|
|
store = seed_store_from_bundle(str(_BUNDLE_DIR))
|
|
query = bundle_candidate_features(str(_BUNDLE_DIR))
|
|
hits = store.retrieve(query, k=3)
|
|
assert len(hits) == 1
|
|
assert "0.82" in hits[0].rationale
|
|
|
|
|
|
# --- S2.5 (Step 7): inbox herding — vocabulary SKIP + rationale-cap (skip) + file-count (fail-fast) ---
|
|
|
|
|
|
def _feats(code: str, magnitude: float = 1.0) -> ProposalFeatures:
|
|
return ProposalFeatures(
|
|
affected_codes=frozenset({code}), measure_type="m", claimed_saving_nok=magnitude
|
|
)
|
|
|
|
|
|
def test_load_skips_unknown_decision_vocabulary(tmp_path) -> None:
|
|
"""T-2.5a: an inbox verdict whose decision is outside the binary run-path vocabulary
|
|
{approved, rejected} is SKIPPED (never enters the store) — tolerant load, never raises. Detach
|
|
the vocabulary check → the ``banana`` verdict enters the store → RED."""
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("X"), "banana", "weird decision"))
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("Y", 2.0), "approved", "fine"))
|
|
|
|
decisions = [v.decision for v in load_verdicts_from_dir(str(tmp_path))]
|
|
assert "banana" not in decisions
|
|
assert "approved" in decisions
|
|
|
|
|
|
def test_load_skips_oversized_rationale_with_log(tmp_path, caplog) -> None:
|
|
"""T-2.5b: a rationale over ``max_rationale_len`` is SKIPPED and logged (caplog-observable) — a
|
|
per-file tolerant skip, NOT a raise (contrast the file-count cap)."""
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("X"), "approved", "x" * 500))
|
|
|
|
with caplog.at_level(logging.WARNING):
|
|
loaded = load_verdicts_from_dir(str(tmp_path), max_rationale_len=100)
|
|
|
|
assert loaded == []
|
|
assert any("rationale" in r.message.lower() for r in caplog.records)
|
|
|
|
|
|
def test_load_fails_fast_over_max_files(tmp_path) -> None:
|
|
"""T-2.5c: a file count over ``max_files`` per merge is a fail-fast RAISE (aggregate guard,
|
|
contrast the per-file tolerant skips)."""
|
|
for i in range(3):
|
|
write_verdict(
|
|
str(tmp_path), capture_verdict(_feats(f"C{i}", float(i + 1)), "approved", "r")
|
|
)
|
|
|
|
with pytest.raises(ValueError):
|
|
load_verdicts_from_dir(str(tmp_path), max_files=2)
|
|
|
|
|
|
def test_load_within_caps_is_unchanged(tmp_path) -> None:
|
|
"""Control: with vocabulary-valid decisions and no cap breach, the loader behaves exactly as
|
|
before — the tolerant contract (which the Step-7 loop relies on) is intact."""
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("X"), "approved", "ok"))
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("Y", 2.0), "rejected", "no"))
|
|
|
|
loaded = load_verdicts_from_dir(str(tmp_path), max_rationale_len=100, max_files=10)
|
|
assert {v.decision for v in loaded} == {"approved", "rejected"}
|
|
|
|
|
|
def test_load_skips_non_utf8_file(tmp_path) -> None:
|
|
"""T-2.5d: a hand-authored inbox verdict saved in Latin-1 (Norwegian ``æ/ø/å``) is valid JSON bytes
|
|
but INVALID UTF-8; it matches the ``*.json`` glob yet must be SKIPPED, never raise — the loader's
|
|
per-file tolerant contract (the Step-7 loop relies on it). Detach the ``UnicodeDecodeError`` catch
|
|
→ ``read_text(encoding='utf-8')`` raises → RED. The valid verdict alongside still loads."""
|
|
(tmp_path / "latin1.json").write_bytes(
|
|
json.dumps(
|
|
{
|
|
"id": "X",
|
|
"decision": "approved",
|
|
"rationale": "godkjent på møtet",
|
|
"proposal_features": {
|
|
"affected_codes": ["05.2"],
|
|
"measure_type": "m",
|
|
"claimed_saving_nok": 200.0,
|
|
},
|
|
},
|
|
ensure_ascii=False,
|
|
).encode("latin-1")
|
|
)
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("Y"), "approved", "fine"))
|
|
|
|
loaded = load_verdicts_from_dir(str(tmp_path))
|
|
assert [v.decision for v in loaded] == ["approved"] # latin1 file skipped, the valid one loads
|
|
|
|
|
|
def test_load_skips_non_iterable_affected_codes(tmp_path) -> None:
|
|
"""Parity ground-truth: ``verdict_from_dict``'s ``frozenset(affected_codes)`` raises ``TypeError``
|
|
on a non-iterable (``null``), and ``load_verdicts_from_dir`` SKIPS it. This pins the reference the
|
|
hitl inbox predicate mirrors (``test_hitl_loadbearing.test_inbox_idset_skips_non_iterable_affected_codes``)
|
|
so the parity contract cannot silently rot on the loader side."""
|
|
(tmp_path / "bad.json").write_text(
|
|
json.dumps(
|
|
{
|
|
"id": "X",
|
|
"decision": "approved",
|
|
"rationale": "r",
|
|
"proposal_features": {
|
|
"affected_codes": None,
|
|
"measure_type": "m",
|
|
"claimed_saving_nok": 1.0,
|
|
},
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
write_verdict(str(tmp_path), capture_verdict(_feats("Y"), "approved", "fine"))
|
|
|
|
loaded = load_verdicts_from_dir(str(tmp_path))
|
|
assert [v.decision for v in loaded] == ["approved"] # non-iterable affected_codes skipped
|
|
|
|
|
|
def test_no_inbox_json_uses_approved_with_adjustment() -> None:
|
|
"""Assumption 3 (TDD guard): no shipped/test ``.json`` uses the promotion-only decision
|
|
``approved_with_adjustment`` — the vocabulary SKIP would now silently drop it. It lives only in
|
|
bundle-seed frontmatter + the promotion gate ``_APPROVED_DECISIONS``, never a binary run-path
|
|
inbox file."""
|
|
root = Path(__file__).resolve().parents[1]
|
|
offenders = [
|
|
p
|
|
for p in root.rglob("*.json")
|
|
if ".venv" not in p.parts
|
|
and ".git" not in p.parts
|
|
and "approved_with_adjustment" in p.read_text(encoding="utf-8", errors="ignore")
|
|
]
|
|
assert offenders == [], f"inbox JSON with promotion-only decision: {offenders}"
|