portfolio-optimiser/tests/test_mandate_run_loadbearing.py
Kjell Tore Guttormsen 938a1ca30e feat(row6): a proposal whose approach declared no requirement is unsupported
Stress round 6 validated three falsification arms, and every validated
approach rested only on run-level declarations nobody can attribute to one
approach. declare_requirement now takes a required approach_id (a mandate
id or own-proposal; an unknown id is refused naming the valid ones), and a
ValidatedProposal whose approach has neither a mandate requirement nor a
declaration under its own id becomes validator.Unsupported - a Rejection
subclass carrying the validator's own ruling, reported as `unsupported` in
coverage, the outcome artefact, the settlement and the judge, and never
counted or summed. The rule is active whenever the debate held the
declaration tool, the micro base included; the road and pre-pass paths are
untouched. Declaration quality is not judged, so the rule can be satisfied
by declaring any document the run read.

The v1 gate's row 6 probes pass; its artefact half reads IKKE MÅLT because
stress round 6 predates approach-addressed declarations, and IKKE MÅLT is
never green - it fails the exit code.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-17 16:40:54 +02:00

209 lines
9.1 KiB
Python

"""Load-bearing: a run EVALUATES every commissioned approach, and REPORTS what happened to each
(Trekk A3/A4 — krav 1 and 2).
Krav 1 is only half-met by carrying an approach into the prompt (``test_mandate_generation_
loadbearing.py``): the expert asked for their approaches to be *concretely evaluated*, which means
each one must reach a verdict and each verdict must be visible. The defect class this pins is
silence — an approach that was commissioned but never evaluated, or evaluated and quietly dropped
because a different one won, is indistinguishable from one that was never ordered.
Detach points, each RED on its own:
* run one generation instead of one per approach -> the coverage report loses rows;
* drop the rejected row (report only what validated) -> the expert's approach vanishes silently;
* drop the own-proposal row -> "og/eller" becomes "or".
The control (``test_no_mandate_runs_exactly_one_generation_with_empty_coverage``) proves the whole
addition is inert without a mandate: today's single-shot path, unchanged.
"""
from __future__ import annotations
from collections.abc import Callable
from pathlib import Path
import pytest
from portfolio_optimiser.mandate import BindingRequirement, OWN_PROPOSAL_ID, Approach, Mandate
from portfolio_optimiser.simulation import ScriptedChatClient
from portfolio_optimiser.run import run_project
from portfolio_optimiser.validator import Rejection, ValidatedProposal
from portfolio_optimiser.verdicts import VerdictStore
BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"}
# BYGG-KONTOR-NORD: affected total = 300000 x 1.0 -> degenerate Monte Carlo P90 = 0.30 x 300000
# = 90000. A claim <= 90000 validates; a claim above it is REJECTED by the deterministic validator.
def _reply(measure: str, claimed: int) -> str:
return (
f'{{"measure":"{measure}","affected_items":'
f'[{{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}}],'
f'"claimed_saving_nok":{claimed}}}'
)
# Labels are chosen to be ABSENT from the bundle's own prose: "LED-retrofit" appears in 6 of the
# bundle's files, so a client keyed on it would match every prompt through the context and prove
# nothing about which approach was bound to which call.
#: Row 6: a commissioned approach validates only on a requirement of its own. These tests are about
#: selection and per-approach artefacts, not about declarations, so the commission names one.
_REQ = BindingRequirement(path="tiltak-led-retrofit.md", ref="Krav 1")
_LED = Approach(id="led-retrofit", label="Behovsstyrt belysning i fellesarealer", requirement=_REQ)
_HVAC = Approach(id="hvac-swap", label="Utskifting av ventilasjonsaggregat", requirement=_REQ)
#: LED validates (30k <= cap); HVAC is above the cap -> the validator rejects it.
_REPLY_BY_LABEL = {
_LED.label: _reply("Behovsstyrt belysning i fellesarealer", 30_000),
_HVAC.label: _reply("Utskifting av ventilasjonsaggregat", 200_000),
}
_DEFAULT_REPLY = _reply("Systemets eget forslag", 20_000)
def _select_reply(blob: str, _role: str) -> str:
"""Reply according to WHICH approach the prompt carries — so a per-approach outcome can only
differ if the loop really bound that approach to that call. Plugged into the CANONICAL
``ScriptedChatClient`` selector seam rather than a copied ``_inner_get_response`` body (S2.5
consolidation guard)."""
return next((r for label, r in _REPLY_BY_LABEL.items() if label in blob), _DEFAULT_REPLY)
def _factory(sink: list[str]) -> Callable[[str], ScriptedChatClient]:
def factory(role: str) -> ScriptedChatClient:
return ScriptedChatClient(
sink=sink, role=role, reply_selector=_select_reply, default_reply=_DEFAULT_REPLY
)
return factory
def _generation_prompts(sink: list[str]) -> list[str]:
"""Generation-call prompts only (``_build_messages`` embeds 'SavingsProposal'), isolated from
the debate-round prompts sharing the sink."""
return [p for p in sink if "SavingsProposal" in p]
async def _run(mandate: Mandate | None, sink: list[str]):
return await run_project(
"BYGG-KONTOR-NORD",
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=VerdictStore(verdicts=[]),
client_factory=_factory(sink),
mandate=mandate,
)
def _row(result, approach_id: str):
return next((c for c in result.coverage if c.id == approach_id), None)
async def test_every_commissioned_approach_is_evaluated() -> None:
"""Both commissioned approaches are generated for — one prompt each, each bound to its own
approach. One generation for two approaches means only one was ever evaluated."""
sink: list[str] = []
mandate = Mandate(
objective="Cut energy cost without rebuilding.",
approaches=(_LED, _HVAC),
allow_own_proposals=False,
)
result = await _run(mandate, sink)
prompts = _generation_prompts(sink)
assert sum(_LED.label in p for p in prompts) >= 1
assert sum(_HVAC.label in p for p in prompts) >= 1
assert {c.id for c in result.coverage} == {"led-retrofit", "hvac-swap"}
async def test_a_rejected_approach_is_reported_not_dropped() -> None:
"""The approach the validator rejected is STILL on the report, with the reason — this is the
silence the coverage report exists to prevent."""
sink: list[str] = []
mandate = Mandate(
objective="Cut energy cost without rebuilding.",
approaches=(_LED, _HVAC),
allow_own_proposals=False,
)
result = await _run(mandate, sink)
led, hvac = _row(result, "led-retrofit"), _row(result, "hvac-swap")
assert led is not None and hvac is not None
assert led.status == "validated"
assert hvac.status == "rejected"
assert hvac.detail, "a rejected approach must carry the validator's reason, not a bare status"
async def test_own_proposal_is_reported_alongside_commissioned_ones() -> None:
"""'og/eller': with ``allow_own_proposals`` the system's own candidate is evaluated too, and
reported under its own reserved row rather than merged into an expert's."""
sink: list[str] = []
mandate = Mandate(
objective="Cut energy cost without rebuilding.",
approaches=(_LED,),
allow_own_proposals=True,
)
result = await _run(mandate, sink)
assert {c.id for c in result.coverage} == {"led-retrofit", OWN_PROPOSAL_ID}
@pytest.mark.parametrize("big_first", [True, False])
async def test_outcome_is_the_best_validated_candidate_deterministically(big_first: bool) -> None:
"""With more than one validated approach the run still returns ONE outcome (portfolio
aggregation, outbox and HITL keying rest on that), chosen by highest validated saving.
BOTH orderings are exercised, and that is the whole point: a first run of this test placed the
bigger approach LAST, where "highest saving" and "whichever ran last" give the same answer — an
order-dependent implementation passed it (MEASURED: the mutation stayed green). A test whose
scenario cannot separate the two implementations proves nothing about either. With both
orderings pinned, ``produced[-1]`` fails one case and ``produced[0]`` fails the other.
"""
sink: list[str] = []
big = Approach(
id="big", label="Behovsstyrt belysning i fellesarealer", requirement=_REQ
) # 30k, validates
small = Approach(
id="small", label="Nattsenking av temperatur", requirement=_REQ
) # default reply, 20k
mandate = Mandate(
objective="Cut energy cost without rebuilding.",
approaches=(big, small) if big_first else (small, big),
allow_own_proposals=False,
)
result = await _run(mandate, sink)
assert isinstance(result.outcome, ValidatedProposal)
assert result.outcome.proposal.claimed_saving_nok == 30_000
assert all(c.status == "validated" for c in result.coverage)
async def test_all_rejected_still_returns_a_rejection_and_full_coverage() -> None:
"""When nothing validates the run still reports every approach — and the outcome stays a
typed ``Rejection`` (never an empty or fabricated success)."""
sink: list[str] = []
hvac2 = Approach(id="hvac-2", label="Utskifting av ventilasjonsaggregat")
mandate = Mandate(
objective="Cut energy cost without rebuilding.",
approaches=(_HVAC, hvac2),
allow_own_proposals=False,
)
result = await _run(mandate, sink)
assert isinstance(result.outcome, Rejection)
assert {c.id for c in result.coverage} == {"hvac-swap", "hvac-2"}
assert all(c.status == "rejected" for c in result.coverage)
async def test_no_mandate_runs_exactly_one_generation_with_empty_coverage() -> None:
"""CONTROL: without a mandate the run is byte-for-byte the pre-Trekk-A one — a single
generation, and no coverage report claiming approaches nobody commissioned."""
sink: list[str] = []
result = await _run(None, sink)
assert len(_generation_prompts(sink)) == 1
assert result.coverage == ()
assert isinstance(result.outcome, ValidatedProposal)