feat(validator): anchor the deterministic gate to the project's real cost baseline (S4.0)
Every stage of validate_proposal reasoned only about numbers the proposal itself supplied, so an internally-consistent hallucination cleared the whole gate (F3). A new stage 0 reconciles each affected_item against the project's CostBaseline before the CBC solve: an unknown cost code is rejected, and a real code carrying a quantity/unit_cost outside the configured tolerance (5% default, relative to the baseline value) is rejected. Validation, never repair. The baseline argument is OPTIONAL (None = pre-S4.0 behaviour), but both run paths set it: the road path projects project.cost_items, the bundle path loads cost-baseline.json when the bundle ships one. Bundles written before the amendment stay un-anchored, so the commons-owned goldens run byte-identically; a baseline that exists but is malformed still raises on both loaders. F8: the method-specific cap now comes from the METHOD_CAPS registry (measure type -> fraction, injectable) instead of an energy_efficiency string comparison. The baseline format and tolerance semantics were decided locally — the commons amendment (D-A pt. 2) never arrived, exactly as in S3.2. D7 mirroring stays open. Three portfolio fixtures quoted cost codes belonging to OTHER projects; the new gate caught them. They now quote each project's own lines, and the two copied REPLIES tables import the single source instead of drifting from it. Load-bearing measured (tests/test_s40_cost_baseline_loadbearing.py), six mutations all red: detach the reconciliation stage; detach the magnitude tolerance; detach the road wiring; detach the bundle wiring; ignore the injected cap registry; make the optional loader tolerant of malformed content. Control: with the road wiring detached the repaired portfolio fixtures still pass, so they are not masking the seam. 597 -> 612 tests. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01JdwK7bQ4BZkWH4t8MRDKb4
This commit is contained in:
parent
012adc0a3c
commit
126807aee7
16 changed files with 645 additions and 69 deletions
|
|
@ -8,11 +8,13 @@ the gated live arm (Step 14).
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections.abc import Callable, Sequence
|
||||
|
||||
import pytest
|
||||
from agent_framework import BaseChatClient
|
||||
|
||||
from portfolio_optimiser.reference_domain import load_reference_projects
|
||||
from portfolio_optimiser.simulation import ScriptedChatClient
|
||||
from portfolio_optimiser.verdicts import VerdictStore, seed_store
|
||||
|
||||
|
|
@ -57,29 +59,63 @@ def make_client_factory() -> Callable[..., Callable[[str], BaseChatClient]]:
|
|||
return _make
|
||||
|
||||
|
||||
# A generic VALID SavingsProposal reply for any project not present in a portfolio reply map:
|
||||
# affected total = 1 x 100_000 = 100_000, P90 = 0.30 x 100_000 = 30_000, claimed 20_000 <= both
|
||||
# (Pydantic affected-total invariant and the validator P90 gate) -> always validates.
|
||||
_DEFAULT_CLAIM = 20_000
|
||||
# Last-resort reply for a prompt naming NO known reference project (the anchored per-project
|
||||
# fallback below cannot be built then). Kept for that case only.
|
||||
_PORTFOLIO_DEFAULT_REPLY = (
|
||||
'{"measure":"Reduce scope","affected_items":'
|
||||
'[{"code":"01.1","quantity":1,"unit_cost":100000}],"claimed_saving_nok":20000}'
|
||||
)
|
||||
|
||||
|
||||
def _anchored_default_replies() -> dict[str, str]:
|
||||
"""A VALID default proposal PER reference project, quoting that project's OWN first cost line
|
||||
verbatim (S4.0): since the road path anchors the validator to ``project.cost_items``, a generic
|
||||
reply carrying an invented magnitude for code ``01.1`` is now — correctly — rejected as a
|
||||
fabricated cost line. Anchoring the fixture is the fix; weakening the gate is not.
|
||||
|
||||
``claimed_saving_nok`` stays ``20_000`` for every project, exactly as the single generic reply
|
||||
claimed before, so every ledger/goal/budget assertion built on that figure is unchanged. Each
|
||||
project's first line is ``01.1 Rigg og drift`` at >= 480 000 NOK, so P90 (>= 144 000) clears the
|
||||
claim on every project."""
|
||||
replies: dict[str, str] = {}
|
||||
for project in load_reference_projects():
|
||||
line = project.cost_items[0]
|
||||
replies[project.id] = json.dumps(
|
||||
{
|
||||
"measure": "Reduce scope",
|
||||
"affected_items": [
|
||||
{"code": line.code, "quantity": line.quantity, "unit_cost": line.unit_cost}
|
||||
],
|
||||
"claimed_saving_nok": _DEFAULT_CLAIM,
|
||||
}
|
||||
)
|
||||
return replies
|
||||
|
||||
|
||||
class _ProjectAwareUsageChatClient(ScriptedChatClient):
|
||||
"""Selects its reply by scanning the incoming prompt for a known ``project_id`` substring (the
|
||||
prompt embeds ``project.id`` at run.py:162 and generate.py:48), falling back to a default valid
|
||||
proposal — so ``run_portfolio``'s single ``client_factory`` stays production-shaped while tests
|
||||
vary the proposal per project. A THIN subclass: the prompt-scan lives in its selector, the shared
|
||||
``_inner_get_response`` body in the canonical."""
|
||||
``_inner_get_response`` body in the canonical.
|
||||
|
||||
The fallback is itself project-aware (S4.0): a prompt naming a reference project gets that
|
||||
project's baseline-anchored default reply, so an un-mapped project still produces a proposal the
|
||||
anchored validator admits. Only a prompt naming NO known project falls through to
|
||||
``default_reply``."""
|
||||
|
||||
def __init__(
|
||||
self, replies: dict[str, str], *, default_reply: str, tokens_per_reply: int = 8
|
||||
) -> None:
|
||||
table = dict(replies)
|
||||
anchored = _anchored_default_replies()
|
||||
|
||||
def _select(blob: str, _role: str) -> str:
|
||||
return next((r for pid, r in table.items() if pid in blob), default_reply)
|
||||
explicit = next((r for pid, r in table.items() if pid in blob), None)
|
||||
if explicit is not None:
|
||||
return explicit
|
||||
return next((r for pid, r in anchored.items() if pid in blob), default_reply)
|
||||
|
||||
super().__init__(
|
||||
reply_selector=_select, default_reply=default_reply, tokens_per_reply=tokens_per_reply
|
||||
|
|
|
|||
|
|
@ -28,29 +28,34 @@ _OFFLINE_MODEL_MAP = {
|
|||
|
||||
# The synthetic reply IS the proposal: generate._parse_ir builds affected_items (each with its
|
||||
# own quantity/unit_cost) straight from this JSON, and the validator's P90 = 0.30 x Σ(qty·unit_cost)
|
||||
# ONLY when ``assumptions`` is empty (degenerate Monte Carlo, validator.py:108-113). All three
|
||||
# replies therefore OMIT ``assumptions`` and carry explicit magnitudes so each
|
||||
# ``claimed_saving_nok`` <= P90. Verified against validator + ir:
|
||||
# FV42-GSV-E1 Σ=1,482,500 P90=444,750 claimed 200,000 -> validates
|
||||
# RV13-RAS-TP Σ= 756,000 P90=226,800 claimed 130,000 -> validates (decoy)
|
||||
# BRU-LAKS-REHAB Σ=2,580,500 P90=774,150 claimed 210,000 -> validates
|
||||
# ONLY when ``assumptions`` is empty (degenerate Monte Carlo, validator.py). All three replies
|
||||
# therefore OMIT ``assumptions`` and carry explicit magnitudes so each ``claimed_saving_nok`` <= P90.
|
||||
#
|
||||
# S4.0: every line below quotes a cost line the project ACTUALLY has, verbatim from
|
||||
# reference_projects.json — the road path now anchors the validator to ``project.cost_items``, so a
|
||||
# reply quoting another project's code (which these fixtures used to do) is rejected as a fabricated
|
||||
# cost line. Verified against reference_projects.json + validator + ir:
|
||||
# FV42-GSV-E1 01.1 1x850,000 + 05.2 4300x215 Σ=1,774,500 P90=532,350 claimed 200,000
|
||||
# RV13-RAS-TP 22.4 610x3,850 Σ=2,348,500 P90=704,550 claimed 130,000 (decoy)
|
||||
# BRU-LAKS-REHAB 01.1 1x620,000 + 87.3 640x980 Σ=1,247,200 P90=374,160 claimed 210,000
|
||||
# ``measure`` is byte-identical "Reduce scope" for FV42+BRU (measure-match is exact string
|
||||
# equality, verdicts.py:68) and "Material substitution" for the decoy, so the BRU<->FV42 pair
|
||||
# overlaps (shared code 05.2 + measure + magnitude bucket) while the decoy does not.
|
||||
# equality, verdicts.py:68) and "Material substitution" for the decoy, so the BRU<->FV42 pair still
|
||||
# overlaps (shared code 01.1 + measure + magnitude bucket) while the decoy does not. 01.1 replaces
|
||||
# 05.2 as the shared code because it is the only code both projects genuinely carry.
|
||||
REPLIES = {
|
||||
"FV42-GSV-E1": (
|
||||
'{"measure":"Reduce scope","affected_items":['
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"03.1","quantity":1800,"unit_cost":310}],"claimed_saving_nok":200000}'
|
||||
'{"code":"01.1","quantity":1,"unit_cost":850000},'
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215}],"claimed_saving_nok":200000}'
|
||||
),
|
||||
"RV13-RAS-TP": (
|
||||
'{"measure":"Material substitution","affected_items":['
|
||||
'{"code":"88.2","quantity":180,"unit_cost":4200}],"claimed_saving_nok":130000}'
|
||||
'{"code":"22.4","quantity":610,"unit_cost":3850}],"claimed_saving_nok":130000}'
|
||||
),
|
||||
"BRU-LAKS-REHAB": (
|
||||
'{"measure":"Reduce scope","affected_items":['
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"07.4","quantity":2400,"unit_cost":690}],"claimed_saving_nok":210000}'
|
||||
'{"code":"01.1","quantity":1,"unit_cost":620000},'
|
||||
'{"code":"87.3","quantity":640,"unit_cost":980}],"claimed_saving_nok":210000}'
|
||||
),
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ from pathlib import Path
|
|||
|
||||
import pytest
|
||||
from agent_framework import Agent
|
||||
from test_portfolio import REPLIES
|
||||
|
||||
from portfolio_optimiser.budget import (
|
||||
Budget,
|
||||
|
|
@ -42,25 +43,11 @@ from portfolio_optimiser.run import run_portfolio
|
|||
|
||||
_PORTFOLIO_IDS = ["FV42-GSV-E1", "RV13-RAS-TP", "BRU-LAKS-REHAB"]
|
||||
|
||||
# Per-project replies (the tested constants of tests/test_portfolio.py, unchanged): all three
|
||||
# validate, and their claimed savings are DISTINCT — which is how a run is keyed back to its
|
||||
# project here, since ``RunResult`` carries no project id. 200000 + 130000 = the first two.
|
||||
REPLIES = {
|
||||
"FV42-GSV-E1": (
|
||||
'{"measure":"Reduce scope","affected_items":['
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"03.1","quantity":1800,"unit_cost":310}],"claimed_saving_nok":200000}'
|
||||
),
|
||||
"RV13-RAS-TP": (
|
||||
'{"measure":"Material substitution","affected_items":['
|
||||
'{"code":"88.2","quantity":180,"unit_cost":4200}],"claimed_saving_nok":130000}'
|
||||
),
|
||||
"BRU-LAKS-REHAB": (
|
||||
'{"measure":"Reduce scope","affected_items":['
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"07.4","quantity":2400,"unit_cost":690}],"claimed_saving_nok":210000}'
|
||||
),
|
||||
}
|
||||
# Per-project replies: IMPORTED from tests/test_portfolio.py rather than copied. The copy claimed to
|
||||
# be "the tested constants ... unchanged" and then drifted — S4.0's baseline anchoring caught it,
|
||||
# because the copies quoted cost codes belonging to OTHER projects. All three validate, and their
|
||||
# claimed savings are DISTINCT — which is how a run is keyed back to its project here, since
|
||||
# ``RunResult`` carries no project id. 200000 + 130000 = the first two.
|
||||
_FIRST_TWO_SAVING = 330000
|
||||
|
||||
# Measured: 32 tokens per run at tokens=8. 80 total funds two runs (32 + 32 = 64) and leaves 16,
|
||||
|
|
|
|||
|
|
@ -34,6 +34,8 @@ from __future__ import annotations
|
|||
import asyncio
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
from test_portfolio import REPLIES
|
||||
|
||||
from portfolio_optimiser.budget import PortfolioBudget, PortfolioMeter
|
||||
from portfolio_optimiser.run import RunFailure, run_portfolio
|
||||
from portfolio_optimiser.simulation import ScriptedChatClient
|
||||
|
|
@ -49,22 +51,11 @@ _DEFAULT_REPLY = (
|
|||
'[{"code":"01.1","quantity":1,"unit_cost":100000}],"claimed_saving_nok":20000}'
|
||||
)
|
||||
|
||||
REPLIES = {
|
||||
"FV42-GSV-E1": (
|
||||
'{"measure":"Reduce scope","affected_items":['
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"03.1","quantity":1800,"unit_cost":310}],"claimed_saving_nok":200000}'
|
||||
),
|
||||
"RV13-RAS-TP": (
|
||||
'{"measure":"Material substitution","affected_items":['
|
||||
'{"code":"88.2","quantity":180,"unit_cost":4200}],"claimed_saving_nok":130000}'
|
||||
),
|
||||
"BRU-LAKS-REHAB": (
|
||||
'{"measure":"Reduce scope","affected_items":['
|
||||
'{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"07.4","quantity":2400,"unit_cost":690}],"claimed_saving_nok":210000}'
|
||||
),
|
||||
}
|
||||
# ``REPLIES`` is IMPORTED from tests/test_portfolio.py (see the import above) rather than copied —
|
||||
# the local copy had drifted onto other projects' cost codes, which S4.0's baseline anchoring
|
||||
# rejects. ``_DEFAULT_REPLY`` above is reached only by a prompt naming none of the three mapped
|
||||
# projects; on the anchored road path such a reply is rejected as a fabricated cost line, which is
|
||||
# the correct outcome for a project this fixture never described.
|
||||
|
||||
# Measured in tests/test_portfolio_budget_loadbearing.py: 4 chat calls x ``tokens`` per reply, so a
|
||||
# completed run costs a flat 32 tokens at tokens=8.
|
||||
|
|
|
|||
285
tests/test_s40_cost_baseline_loadbearing.py
Normal file
285
tests/test_s40_cost_baseline_loadbearing.py
Normal file
|
|
@ -0,0 +1,285 @@
|
|||
"""S4.0 load-bearing seam: the deterministic gate is ANCHORED to the project's real cost baseline.
|
||||
|
||||
Review finding F3: every stage of ``validate_proposal`` reasoned about the numbers the *proposal
|
||||
itself* supplied, so a hallucinated cost line (an invented code, or a real code at an invented
|
||||
magnitude) could clear the whole gate as long as its own arithmetic was internally consistent. The
|
||||
reconciliation stage closes that: each ``affected_item`` must correspond to a line in the project's
|
||||
cost baseline, within a configured tolerance.
|
||||
|
||||
Every RED here is a genuine OUTCOME FLIP, not a reason-string check: the fabricated proposals are
|
||||
deliberately built to pass the P90 / nominal / method stages, so detaching the reconciliation makes
|
||||
them ``ValidatedProposal`` again. Controls prove causality (a real baseline line, same shape,
|
||||
validates), and the no-baseline arm proves the argument stays OPTIONAL (``None`` = pre-S4.0
|
||||
behaviour, which is why the existing suite stands).
|
||||
|
||||
Measured detach points (see the session log): the reconciliation stage · the magnitude tolerance ·
|
||||
the road-path wiring in ``run.py`` · the bundle-path wiring · the method-cap registry key (F8).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from conftest import SyntheticUsageChatClient
|
||||
from pydantic import ValidationError
|
||||
|
||||
from portfolio_optimiser import okf
|
||||
from portfolio_optimiser.ir import AffectedItem, CostBaseline, CostBaselineLine, SavingsProposal
|
||||
from portfolio_optimiser.reference_domain import load_reference_projects
|
||||
from portfolio_optimiser.run import run_project
|
||||
from portfolio_optimiser.validator import (
|
||||
Rejection,
|
||||
ValidatedProposal,
|
||||
baseline_from_project,
|
||||
validate_proposal,
|
||||
)
|
||||
|
||||
# The repo-local S4.0 fixture bundle: the ONLY bundle carrying a ``cost-baseline.json`` (the pre-
|
||||
# amendment bundles deliberately have none — that is the optional-argument control below).
|
||||
_DATA = Path(__file__).resolve().parents[1] / "src" / "portfolio_optimiser" / "data" / "bundles"
|
||||
BASELINE_BUNDLE = _DATA / "bygg-energi-baseline-mikro"
|
||||
PRE_AMENDMENT_BUNDLE = _DATA / "bygg-energi-mikro-a"
|
||||
|
||||
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"}
|
||||
|
||||
# FV42-GSV-E1's real cost line 05.2 (Asfalt Ab11): 4300 m2 x 215 NOK. Affected total 924500 ->
|
||||
# degenerate P90 = 0.30 x 924500 = 277350, so claimed 200000 clears every pre-S4.0 stage.
|
||||
_REAL_CODE = "05.2"
|
||||
_REAL_QTY = 4300.0
|
||||
_REAL_UNIT_COST = 215.0
|
||||
_REAL_CLAIM = 200000.0
|
||||
|
||||
# The F3 scenario: an invented code carrying a 10 MNOK line. The claim is set at exactly the generic
|
||||
# feasible (0.30 x 10 MNOK) so the fabrication is numerically IMPECCABLE — every pre-S4.0 stage
|
||||
# passes it. Only the baseline reconciliation can reject it, which is what makes the detach a flip.
|
||||
_FAKE_CODE = "XX"
|
||||
_FAKE_UNIT_COST = 10_000_000.0
|
||||
_FAKE_CLAIM = 3_000_000.0
|
||||
|
||||
|
||||
def _fv42_baseline() -> CostBaseline:
|
||||
return baseline_from_project(
|
||||
next(p for p in load_reference_projects() if p.id == "FV42-GSV-E1")
|
||||
)
|
||||
|
||||
|
||||
def _proposal(code: str, quantity: float, unit_cost: float, claimed: float) -> SavingsProposal:
|
||||
return SavingsProposal(
|
||||
project_id="FV42-GSV-E1",
|
||||
measure="Reduce scope",
|
||||
affected_items=[AffectedItem(code=code, quantity=quantity, unit_cost=unit_cost)],
|
||||
claimed_saving_nok=claimed,
|
||||
assumptions={},
|
||||
)
|
||||
|
||||
|
||||
# --- Arm 1: the reconciliation stage itself -------------------------------------------------------
|
||||
|
||||
|
||||
def test_fabricated_cost_code_is_rejected() -> None:
|
||||
"""RED (F3): a proposal citing a cost code that exists nowhere in the project's baseline is
|
||||
rejected, even though its own arithmetic clears the LP/P90/nominal stages. Detach the
|
||||
reconciliation stage and the SAME proposal validates."""
|
||||
result = validate_proposal(
|
||||
_proposal(_FAKE_CODE, 1.0, _FAKE_UNIT_COST, _FAKE_CLAIM), baseline=_fv42_baseline()
|
||||
)
|
||||
assert isinstance(result, Rejection), "a hallucinated cost code must never reach validated"
|
||||
assert "unknown cost code" in result.reason
|
||||
assert _FAKE_CODE in result.reason
|
||||
|
||||
|
||||
def test_real_baseline_line_still_validates() -> None:
|
||||
"""Causality control: the same shape of proposal on a REAL baseline line validates — so the
|
||||
rejection above is caused by the code being absent from the baseline, not by the new stage
|
||||
rejecting everything."""
|
||||
result = validate_proposal(
|
||||
_proposal(_REAL_CODE, _REAL_QTY, _REAL_UNIT_COST, _REAL_CLAIM), baseline=_fv42_baseline()
|
||||
)
|
||||
assert isinstance(result, ValidatedProposal)
|
||||
|
||||
|
||||
def test_baseline_is_optional_and_none_is_pre_s40_behaviour() -> None:
|
||||
"""The baseline argument is OPTIONAL: with ``None`` the fabricated proposal validates exactly as
|
||||
it did before S4.0. This is the property the existing suite rests on — and the reason the RED
|
||||
above is a flip rather than a tightening of an already-rejecting path."""
|
||||
result = validate_proposal(_proposal(_FAKE_CODE, 1.0, _FAKE_UNIT_COST, _FAKE_CLAIM))
|
||||
assert isinstance(result, ValidatedProposal)
|
||||
|
||||
|
||||
# --- Arm 2: the magnitude tolerance ---------------------------------------------------------------
|
||||
|
||||
|
||||
def test_inflated_unit_cost_on_a_real_code_is_rejected() -> None:
|
||||
"""RED: a REAL cost code at an invented unit_cost (+20%, well past the 5% default tolerance) is
|
||||
rejected. Detach the tolerance check and only the code-membership test remains — the inflated
|
||||
line then validates, because the code itself is genuine."""
|
||||
inflated = _REAL_UNIT_COST * 1.20
|
||||
result = validate_proposal(
|
||||
_proposal(_REAL_CODE, _REAL_QTY, inflated, _REAL_CLAIM), baseline=_fv42_baseline()
|
||||
)
|
||||
assert isinstance(result, Rejection)
|
||||
assert "unit_cost" in result.reason and _REAL_CODE in result.reason
|
||||
|
||||
|
||||
def test_inflated_quantity_on_a_real_code_is_rejected() -> None:
|
||||
"""RED: the same for quantity — a real code at an invented quantity (+20%) is rejected."""
|
||||
result = validate_proposal(
|
||||
_proposal(_REAL_CODE, _REAL_QTY * 1.20, _REAL_UNIT_COST, _REAL_CLAIM),
|
||||
baseline=_fv42_baseline(),
|
||||
)
|
||||
assert isinstance(result, Rejection)
|
||||
assert "quantity" in result.reason
|
||||
|
||||
|
||||
def test_within_tolerance_deviation_is_admitted() -> None:
|
||||
"""Causality control for the tolerance: a 2% deviation (rounding-scale, inside the 5% default)
|
||||
validates — the rejections above are caused by the SIZE of the deviation, not by any deviation
|
||||
at all."""
|
||||
result = validate_proposal(
|
||||
_proposal(_REAL_CODE, _REAL_QTY, _REAL_UNIT_COST * 1.02, _REAL_CLAIM),
|
||||
baseline=_fv42_baseline(),
|
||||
)
|
||||
assert isinstance(result, ValidatedProposal)
|
||||
|
||||
|
||||
def test_tolerance_is_configurable() -> None:
|
||||
"""The tolerance is config, not a constant: the same 2% deviation is rejected under a stricter
|
||||
caller-supplied tolerance."""
|
||||
result = validate_proposal(
|
||||
_proposal(_REAL_CODE, _REAL_QTY, _REAL_UNIT_COST * 1.02, _REAL_CLAIM),
|
||||
baseline=_fv42_baseline(),
|
||||
tolerance=0.001,
|
||||
)
|
||||
assert isinstance(result, Rejection)
|
||||
|
||||
|
||||
# --- Arm 3: the loader (fail-fast, mirroring ``load_ir_projection``) -------------------------------
|
||||
|
||||
|
||||
def test_bundle_baseline_loads_from_the_fixture() -> None:
|
||||
baseline = okf.load_cost_baseline(str(BASELINE_BUNDLE))
|
||||
assert baseline.project_id == "BYGG-ENERGI-BASELINE-MIKRO"
|
||||
assert baseline.items["ENERGI-TOTAL-EL"] == CostBaselineLine(quantity=180000, unit_cost=1.0)
|
||||
|
||||
|
||||
def test_missing_baseline_is_fail_fast_but_optional_loader_returns_none() -> None:
|
||||
"""Two deliberately different contracts over the same absence: the fail-fast loader raises (it
|
||||
is authoritative startup input, like ``load_ir_projection``), while the OPTIONAL loader the run
|
||||
path uses returns ``None`` — a bundle written before the amendment is not an error, it is simply
|
||||
un-anchored."""
|
||||
with pytest.raises(FileNotFoundError):
|
||||
okf.load_cost_baseline(str(PRE_AMENDMENT_BUNDLE))
|
||||
assert okf.load_optional_cost_baseline(str(PRE_AMENDMENT_BUNDLE)) is None
|
||||
|
||||
|
||||
def test_malformed_baseline_raises_even_on_the_optional_path(tmp_path) -> None:
|
||||
"""Fail-closed where it matters: a baseline that EXISTS but is malformed raises on BOTH loaders.
|
||||
Tolerating it would silently un-anchor the gate — the RAW-inbox skip rule stops at this layer."""
|
||||
(tmp_path / "cost-baseline.json").write_text(
|
||||
json.dumps({"project_id": "P", "items": {"01.1": {"quantity": 1}}}), encoding="utf-8"
|
||||
)
|
||||
with pytest.raises(ValidationError):
|
||||
okf.load_optional_cost_baseline(str(tmp_path))
|
||||
|
||||
|
||||
# --- Arm 4: the run-path wiring (road + bundle) ---------------------------------------------------
|
||||
|
||||
|
||||
def _reply(code: str, quantity: float, unit_cost: float, claimed: float) -> str:
|
||||
return json.dumps(
|
||||
{
|
||||
"measure": "Reduce scope",
|
||||
"affected_items": [{"code": code, "quantity": quantity, "unit_cost": unit_cost}],
|
||||
"claimed_saving_nok": claimed,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _factory(reply: str):
|
||||
def factory(role: str):
|
||||
return SyntheticUsageChatClient(default_reply=reply)
|
||||
|
||||
return factory
|
||||
|
||||
|
||||
async def test_road_path_anchors_the_gate_to_the_reference_baseline(docs_dir, fresh_store) -> None:
|
||||
"""RED (road wiring): a numerically-impeccable fabricated cost line is REJECTED end-to-end
|
||||
through ``run_project``. Detach the road-path baseline (stop passing it) and the same run
|
||||
returns a ValidatedProposal."""
|
||||
result = await run_project(
|
||||
"FV42-GSV-E1",
|
||||
"local",
|
||||
docs_dir=docs_dir,
|
||||
verdict_input=_VERDICT_INPUT,
|
||||
client_factory=_factory(_reply(_FAKE_CODE, 1.0, _FAKE_UNIT_COST, _FAKE_CLAIM)),
|
||||
store=fresh_store,
|
||||
)
|
||||
assert isinstance(result.outcome, Rejection)
|
||||
assert "unknown cost code" in result.outcome.reason
|
||||
|
||||
|
||||
async def test_road_path_control_real_line_validates(docs_dir, fresh_store) -> None:
|
||||
"""Causality control for the road wiring: the real 05.2 line validates through the same path."""
|
||||
result = await run_project(
|
||||
"FV42-GSV-E1",
|
||||
"local",
|
||||
docs_dir=docs_dir,
|
||||
verdict_input=_VERDICT_INPUT,
|
||||
client_factory=_factory(_reply(_REAL_CODE, _REAL_QTY, _REAL_UNIT_COST, _REAL_CLAIM)),
|
||||
store=fresh_store,
|
||||
)
|
||||
assert isinstance(result.outcome, ValidatedProposal)
|
||||
|
||||
|
||||
async def test_bundle_path_anchors_when_the_bundle_declares_a_baseline(fresh_store) -> None:
|
||||
"""RED (bundle wiring): a bundle that ships ``cost-baseline.json`` anchors its run — the
|
||||
fabricated line is rejected. Detach the bundle-path load and it validates again."""
|
||||
result = await run_project(
|
||||
"BYGG-ENERGI-BASELINE-MIKRO",
|
||||
"local",
|
||||
docs_dir=str(BASELINE_BUNDLE),
|
||||
bundle_dir=str(BASELINE_BUNDLE),
|
||||
verdict_input=_VERDICT_INPUT,
|
||||
client_factory=_factory(_reply(_FAKE_CODE, 1.0, 300000.0, 90000.0)),
|
||||
store=fresh_store,
|
||||
)
|
||||
assert isinstance(result.outcome, Rejection)
|
||||
assert "unknown cost code" in result.outcome.reason
|
||||
|
||||
|
||||
async def test_pre_amendment_bundle_runs_unchanged(fresh_store) -> None:
|
||||
"""Control + backward compatibility: the SAME fabricated reply validates against a bundle with
|
||||
no ``cost-baseline.json``. Anchoring is opt-in per bundle, so every pre-S4.0 bundle (including
|
||||
the commons-owned goldens) runs byte-identically to before."""
|
||||
result = await run_project(
|
||||
"BYGG-ENERGI-MIKRO-A",
|
||||
"local",
|
||||
docs_dir=str(PRE_AMENDMENT_BUNDLE),
|
||||
bundle_dir=str(PRE_AMENDMENT_BUNDLE),
|
||||
verdict_input=_VERDICT_INPUT,
|
||||
client_factory=_factory(_reply(_FAKE_CODE, 1.0, 300000.0, 90000.0)),
|
||||
store=fresh_store,
|
||||
)
|
||||
assert isinstance(result.outcome, ValidatedProposal)
|
||||
|
||||
|
||||
# --- Arm 5: F8 — method caps keyed by config, not the literal measure string ----------------------
|
||||
|
||||
|
||||
def test_method_cap_is_keyed_by_config_not_a_hardcoded_string() -> None:
|
||||
"""F8: the method-specific cap comes from a REGISTRY the caller can supply. A caller-configured
|
||||
cap for a measure with no built-in entry rejects a proposal the generic P90 stage passes — so
|
||||
the rule is keyed by configuration, not by the ``energy_efficiency`` literal."""
|
||||
proposal = SavingsProposal(
|
||||
project_id="P-ASFALT",
|
||||
measure="asfalt_reduction",
|
||||
affected_items=[AffectedItem(code="05.2", quantity=100000, unit_cost=1.0)],
|
||||
claimed_saving_nok=20000,
|
||||
assumptions={},
|
||||
)
|
||||
assert isinstance(validate_proposal(proposal), ValidatedProposal) # generic P90 = 30000
|
||||
capped = validate_proposal(proposal, method_caps={"asfalt_reduction": 0.10})
|
||||
assert isinstance(capped, Rejection)
|
||||
assert "method cap" in capped.reason
|
||||
Loading…
Add table
Add a link
Reference in a new issue