The context sets, the packaged knowledge bases and the example bundles are replaced by one fictitious example set about IT operations in an invented organisation: three context sets (serverrom-2027, driftsavtale-2027 and the two-base drift-og-avtale-2027), two synthetic knowledge bases under src/portfolio_optimiser/data/kunnskapsbaser and two example bundles under src/portfolio_optimiser/data/bundles. Numbers, codes and structural values in tests and fixtures are kept; names, ids and wording change. Dated measurement documents that only recorded runs on the replaced material are deleted. Gate figures measured on the new set are not comparable with earlier ones. The exclusion gate from the previous commit is green: 0 tracked files hit outside the shared/ subtree. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
112 lines
4.1 KiB
Python
112 lines
4.1 KiB
Python
"""Spike D tests — non-tautological top-K retrieval (B2).
|
|
|
|
The true match shares the *structured* similarity fields with the query but uses different
|
|
description text; the decoys share surface description text but differ structurally. A
|
|
text-matching retriever would be fooled by the decoys — a structural one is not.
|
|
Pattern: tests/test_reference_domain.py.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from spikes.d_verdictstore import (
|
|
ExpeLContextProvider,
|
|
ProposalFeatures,
|
|
Verdict,
|
|
VerdictStore,
|
|
seed_store,
|
|
)
|
|
|
|
_QUERY = ProposalFeatures(
|
|
affected_codes=frozenset({"05.2", "03.1"}),
|
|
measure_type="scope_reduction",
|
|
claimed_saving_nok=220_000, # bucket [100k, 500k)
|
|
description="licence scope reduction near head office",
|
|
)
|
|
|
|
|
|
def _store_with_true_match_and_decoys() -> tuple[VerdictStore, str]:
|
|
true_match = Verdict(
|
|
id="TRUE",
|
|
proposal_features=ProposalFeatures(
|
|
affected_codes=frozenset({"05.2", "03.1"}), # same codes
|
|
measure_type="scope_reduction", # same measure type
|
|
claimed_saving_nok=200_000, # same magnitude bucket
|
|
description="zzz totally unrelated wording alpha beta", # DIFFERENT text
|
|
),
|
|
decision="approved",
|
|
rationale="prior scope reduction on the same codes was approved",
|
|
)
|
|
# Decoys share the query's SURFACE text but differ structurally (disjoint codes,
|
|
# different measure type, different magnitude bucket).
|
|
decoy_low = Verdict(
|
|
id="DECOY-LOW",
|
|
proposal_features=ProposalFeatures(
|
|
affected_codes=frozenset({"09.1"}),
|
|
measure_type="rate_renegotiation",
|
|
claimed_saving_nok=50_000, # bucket [0, 100k)
|
|
description="licence scope reduction near head office", # same words as query
|
|
),
|
|
decision="rejected",
|
|
rationale="surface-text decoy",
|
|
)
|
|
decoy_high = Verdict(
|
|
id="DECOY-HIGH",
|
|
proposal_features=ProposalFeatures(
|
|
affected_codes=frozenset({"21.2"}),
|
|
measure_type="material_substitution",
|
|
claimed_saving_nok=700_000, # bucket [500k, 1M)
|
|
description="licence scope reduction extra words",
|
|
),
|
|
decision="rejected",
|
|
rationale="surface-text decoy",
|
|
)
|
|
# Order deliberately not putting the true match first.
|
|
return VerdictStore(verdicts=[decoy_low, true_match, decoy_high]), "TRUE"
|
|
|
|
|
|
def test_retrieve_finds_structural_match_over_text_decoys() -> None:
|
|
store, true_id = _store_with_true_match_and_decoys()
|
|
hits = store.retrieve(_QUERY, k=3)
|
|
assert hits[0].id == true_id # structural match ranks #1 despite different wording
|
|
assert true_id in {h.id for h in hits[:1]} # within top-K (top-1 here)
|
|
|
|
|
|
def test_retrieve_is_deterministic() -> None:
|
|
store, _ = _store_with_true_match_and_decoys()
|
|
assert [h.id for h in store.retrieve(_QUERY, k=3)] == [
|
|
h.id for h in store.retrieve(_QUERY, k=3)
|
|
]
|
|
|
|
|
|
def test_retrieve_rejects_non_positive_k() -> None:
|
|
store, _ = _store_with_true_match_and_decoys()
|
|
with pytest.raises(ValueError):
|
|
store.retrieve(_QUERY, k=0)
|
|
|
|
|
|
def test_seed_store_has_10_to_20_verdicts() -> None:
|
|
store = seed_store()
|
|
assert 10 <= len(store.verdicts) <= 20
|
|
|
|
|
|
class _RecordingContext:
|
|
"""Duck-typed stand-in for SessionContext — records injected instructions without
|
|
depending on the (private) SessionContext internals."""
|
|
|
|
def __init__(self) -> None:
|
|
self.instructions: list[str] = []
|
|
|
|
def extend_instructions(self, items: list[str]) -> None:
|
|
self.instructions.extend(items)
|
|
|
|
|
|
async def test_expel_provider_injects_retrieved_fewshot() -> None:
|
|
store, true_id = _store_with_true_match_and_decoys()
|
|
provider = ExpeLContextProvider(store, _QUERY, k=2)
|
|
# The deterministic deliverable: the few-shot text carries the retrieved true match.
|
|
assert true_id in provider.format_fewshot()
|
|
# The injection hook (real ContextProvider.before_run signature) extends instructions.
|
|
ctx = _RecordingContext()
|
|
await provider.before_run(agent=None, session=None, context=ctx, state={})
|
|
assert len(ctx.instructions) == 1
|
|
assert true_id in ctx.instructions[0]
|