Stress round 6 validated three falsification arms, and every validated approach rested only on run-level declarations nobody can attribute to one approach. declare_requirement now takes a required approach_id (a mandate id or own-proposal; an unknown id is refused naming the valid ones), and a ValidatedProposal whose approach has neither a mandate requirement nor a declaration under its own id becomes validator.Unsupported - a Rejection subclass carrying the validator's own ruling, reported as `unsupported` in coverage, the outcome artefact, the settlement and the judge, and never counted or summed. The rule is active whenever the debate held the declaration tool, the micro base included; the road and pre-pass paths are untouched. Declaration quality is not judged, so the rule can be satisfied by declaring any document the run read. The v1 gate's row 6 probes pass; its artefact half reads IKKE MÅLT because stress round 6 predates approach-addressed declarations, and IKKE MÅLT is never green - it fails the exit code. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
209 lines
9.1 KiB
Python
209 lines
9.1 KiB
Python
"""Load-bearing: a run EVALUATES every commissioned approach, and REPORTS what happened to each
|
|
(Trekk A3/A4 — krav 1 and 2).
|
|
|
|
Krav 1 is only half-met by carrying an approach into the prompt (``test_mandate_generation_
|
|
loadbearing.py``): the expert asked for their approaches to be *concretely evaluated*, which means
|
|
each one must reach a verdict and each verdict must be visible. The defect class this pins is
|
|
silence — an approach that was commissioned but never evaluated, or evaluated and quietly dropped
|
|
because a different one won, is indistinguishable from one that was never ordered.
|
|
|
|
Detach points, each RED on its own:
|
|
|
|
* run one generation instead of one per approach -> the coverage report loses rows;
|
|
* drop the rejected row (report only what validated) -> the expert's approach vanishes silently;
|
|
* drop the own-proposal row -> "og/eller" becomes "or".
|
|
|
|
The control (``test_no_mandate_runs_exactly_one_generation_with_empty_coverage``) proves the whole
|
|
addition is inert without a mandate: today's single-shot path, unchanged.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Callable
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from portfolio_optimiser.mandate import BindingRequirement, OWN_PROPOSAL_ID, Approach, Mandate
|
|
from portfolio_optimiser.simulation import ScriptedChatClient
|
|
from portfolio_optimiser.run import run_project
|
|
from portfolio_optimiser.validator import Rejection, ValidatedProposal
|
|
from portfolio_optimiser.verdicts import VerdictStore
|
|
|
|
BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
|
|
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"}
|
|
|
|
# BYGG-KONTOR-NORD: affected total = 300000 x 1.0 -> degenerate Monte Carlo P90 = 0.30 x 300000
|
|
# = 90000. A claim <= 90000 validates; a claim above it is REJECTED by the deterministic validator.
|
|
|
|
|
|
def _reply(measure: str, claimed: int) -> str:
|
|
return (
|
|
f'{{"measure":"{measure}","affected_items":'
|
|
f'[{{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}}],'
|
|
f'"claimed_saving_nok":{claimed}}}'
|
|
)
|
|
|
|
|
|
# Labels are chosen to be ABSENT from the bundle's own prose: "LED-retrofit" appears in 6 of the
|
|
# bundle's files, so a client keyed on it would match every prompt through the context and prove
|
|
# nothing about which approach was bound to which call.
|
|
#: Row 6: a commissioned approach validates only on a requirement of its own. These tests are about
|
|
#: selection and per-approach artefacts, not about declarations, so the commission names one.
|
|
_REQ = BindingRequirement(path="tiltak-led-retrofit.md", ref="Krav 1")
|
|
_LED = Approach(id="led-retrofit", label="Behovsstyrt belysning i fellesarealer", requirement=_REQ)
|
|
_HVAC = Approach(id="hvac-swap", label="Utskifting av ventilasjonsaggregat", requirement=_REQ)
|
|
|
|
#: LED validates (30k <= cap); HVAC is above the cap -> the validator rejects it.
|
|
_REPLY_BY_LABEL = {
|
|
_LED.label: _reply("Behovsstyrt belysning i fellesarealer", 30_000),
|
|
_HVAC.label: _reply("Utskifting av ventilasjonsaggregat", 200_000),
|
|
}
|
|
_DEFAULT_REPLY = _reply("Systemets eget forslag", 20_000)
|
|
|
|
|
|
def _select_reply(blob: str, _role: str) -> str:
|
|
"""Reply according to WHICH approach the prompt carries — so a per-approach outcome can only
|
|
differ if the loop really bound that approach to that call. Plugged into the CANONICAL
|
|
``ScriptedChatClient`` selector seam rather than a copied ``_inner_get_response`` body (S2.5
|
|
consolidation guard)."""
|
|
return next((r for label, r in _REPLY_BY_LABEL.items() if label in blob), _DEFAULT_REPLY)
|
|
|
|
|
|
def _factory(sink: list[str]) -> Callable[[str], ScriptedChatClient]:
|
|
def factory(role: str) -> ScriptedChatClient:
|
|
return ScriptedChatClient(
|
|
sink=sink, role=role, reply_selector=_select_reply, default_reply=_DEFAULT_REPLY
|
|
)
|
|
|
|
return factory
|
|
|
|
|
|
def _generation_prompts(sink: list[str]) -> list[str]:
|
|
"""Generation-call prompts only (``_build_messages`` embeds 'SavingsProposal'), isolated from
|
|
the debate-round prompts sharing the sink."""
|
|
return [p for p in sink if "SavingsProposal" in p]
|
|
|
|
|
|
async def _run(mandate: Mandate | None, sink: list[str]):
|
|
return await run_project(
|
|
"BYGG-KONTOR-NORD",
|
|
"local",
|
|
docs_dir=str(BUNDLE_DIR),
|
|
bundle_dir=str(BUNDLE_DIR),
|
|
verdict_input=_VERDICT_INPUT,
|
|
store=VerdictStore(verdicts=[]),
|
|
client_factory=_factory(sink),
|
|
mandate=mandate,
|
|
)
|
|
|
|
|
|
def _row(result, approach_id: str):
|
|
return next((c for c in result.coverage if c.id == approach_id), None)
|
|
|
|
|
|
async def test_every_commissioned_approach_is_evaluated() -> None:
|
|
"""Both commissioned approaches are generated for — one prompt each, each bound to its own
|
|
approach. One generation for two approaches means only one was ever evaluated."""
|
|
sink: list[str] = []
|
|
mandate = Mandate(
|
|
objective="Cut energy cost without rebuilding.",
|
|
approaches=(_LED, _HVAC),
|
|
allow_own_proposals=False,
|
|
)
|
|
result = await _run(mandate, sink)
|
|
|
|
prompts = _generation_prompts(sink)
|
|
assert sum(_LED.label in p for p in prompts) >= 1
|
|
assert sum(_HVAC.label in p for p in prompts) >= 1
|
|
assert {c.id for c in result.coverage} == {"led-retrofit", "hvac-swap"}
|
|
|
|
|
|
async def test_a_rejected_approach_is_reported_not_dropped() -> None:
|
|
"""The approach the validator rejected is STILL on the report, with the reason — this is the
|
|
silence the coverage report exists to prevent."""
|
|
sink: list[str] = []
|
|
mandate = Mandate(
|
|
objective="Cut energy cost without rebuilding.",
|
|
approaches=(_LED, _HVAC),
|
|
allow_own_proposals=False,
|
|
)
|
|
result = await _run(mandate, sink)
|
|
|
|
led, hvac = _row(result, "led-retrofit"), _row(result, "hvac-swap")
|
|
assert led is not None and hvac is not None
|
|
assert led.status == "validated"
|
|
assert hvac.status == "rejected"
|
|
assert hvac.detail, "a rejected approach must carry the validator's reason, not a bare status"
|
|
|
|
|
|
async def test_own_proposal_is_reported_alongside_commissioned_ones() -> None:
|
|
"""'og/eller': with ``allow_own_proposals`` the system's own candidate is evaluated too, and
|
|
reported under its own reserved row rather than merged into an expert's."""
|
|
sink: list[str] = []
|
|
mandate = Mandate(
|
|
objective="Cut energy cost without rebuilding.",
|
|
approaches=(_LED,),
|
|
allow_own_proposals=True,
|
|
)
|
|
result = await _run(mandate, sink)
|
|
|
|
assert {c.id for c in result.coverage} == {"led-retrofit", OWN_PROPOSAL_ID}
|
|
|
|
|
|
@pytest.mark.parametrize("big_first", [True, False])
|
|
async def test_outcome_is_the_best_validated_candidate_deterministically(big_first: bool) -> None:
|
|
"""With more than one validated approach the run still returns ONE outcome (portfolio
|
|
aggregation, outbox and HITL keying rest on that), chosen by highest validated saving.
|
|
|
|
BOTH orderings are exercised, and that is the whole point: a first run of this test placed the
|
|
bigger approach LAST, where "highest saving" and "whichever ran last" give the same answer — an
|
|
order-dependent implementation passed it (MEASURED: the mutation stayed green). A test whose
|
|
scenario cannot separate the two implementations proves nothing about either. With both
|
|
orderings pinned, ``produced[-1]`` fails one case and ``produced[0]`` fails the other.
|
|
"""
|
|
sink: list[str] = []
|
|
big = Approach(
|
|
id="big", label="Behovsstyrt belysning i fellesarealer", requirement=_REQ
|
|
) # 30k, validates
|
|
small = Approach(
|
|
id="small", label="Nattsenking av temperatur", requirement=_REQ
|
|
) # default reply, 20k
|
|
mandate = Mandate(
|
|
objective="Cut energy cost without rebuilding.",
|
|
approaches=(big, small) if big_first else (small, big),
|
|
allow_own_proposals=False,
|
|
)
|
|
result = await _run(mandate, sink)
|
|
|
|
assert isinstance(result.outcome, ValidatedProposal)
|
|
assert result.outcome.proposal.claimed_saving_nok == 30_000
|
|
assert all(c.status == "validated" for c in result.coverage)
|
|
|
|
|
|
async def test_all_rejected_still_returns_a_rejection_and_full_coverage() -> None:
|
|
"""When nothing validates the run still reports every approach — and the outcome stays a
|
|
typed ``Rejection`` (never an empty or fabricated success)."""
|
|
sink: list[str] = []
|
|
hvac2 = Approach(id="hvac-2", label="Utskifting av ventilasjonsaggregat")
|
|
mandate = Mandate(
|
|
objective="Cut energy cost without rebuilding.",
|
|
approaches=(_HVAC, hvac2),
|
|
allow_own_proposals=False,
|
|
)
|
|
result = await _run(mandate, sink)
|
|
|
|
assert isinstance(result.outcome, Rejection)
|
|
assert {c.id for c in result.coverage} == {"hvac-swap", "hvac-2"}
|
|
assert all(c.status == "rejected" for c in result.coverage)
|
|
|
|
|
|
async def test_no_mandate_runs_exactly_one_generation_with_empty_coverage() -> None:
|
|
"""CONTROL: without a mandate the run is byte-for-byte the pre-Trekk-A one — a single
|
|
generation, and no coverage report claiming approaches nobody commissioned."""
|
|
sink: list[str] = []
|
|
result = await _run(None, sink)
|
|
|
|
assert len(_generation_prompts(sink)) == 1
|
|
assert result.coverage == ()
|
|
assert isinstance(result.outcome, ValidatedProposal)
|