feat(mandate): a domain expert can commission WHICH approaches a run evaluates
Operator feedback: fagpersoner must be able to name the approaches a run shall
evaluate for a project, and/or ask the system for its own. Today the hypothesis
prompt is hardcoded ("Propose ONE concrete cost-saving measure") and the only
expert-facing lever, --dimension-config, FILTERS what may pass the scoping gate
rather than DIRECTING what is spent attempts on. This is the input that was
missing.
`mandate.py` is the typed commission + a fail-fast loader (mirrors
`load_dimension`/`load_goal_config`): missing or malformed refuses, because a run
must never proceed on a silently degraded commission — the coverage report would
then describe work nobody ordered. Stdlib + pydantic only, so it joins
`_MAF_FREE_MODULES` and can be mirrored to the D7 sibling.
Two refusals carry real defect classes: an EMPTY commission (no approaches and no
own proposals) is a caller error, not a result; and a duplicate approach id — or
one claiming the reserved OWN_PROPOSAL_ID — would collapse two coverage rows onto
one key (the S3.2 key-collision class), which is exactly the silence the coverage
report exists to prevent.
The numeric target is deliberately NOT duplicated here: it already lives in
GoalContract, and two copies of one number drift apart ((p) precedent). The
mandate carries intent; `announce` merely restates the figure.
`_build_messages(approach=...)` switches the opening instruction from *find one*
to *quantify THIS one*, carrying the expert's label and description VERBATIM —
the description is the reason the approach is worth trying, the one part the model
cannot infer from cost data. `approach=None` is byte-identical to the previous
prompt, so every existing run and golden is untouched.
The gate is unmoved: `validate_proposal` is called exactly as before. A
commissioned approach gets no discount — the expert directs what is EVALUATED,
never what is APPROVED.
Load-bearing MEASURED against the whole 695-test suite, four mutations all red:
detach the approach injection (2 red, control stayed green) · let a commissioned
approach bypass the validator · make the mandate loader tolerant · drop the
empty-commission refusal.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01ULCqjLF61rehj5cZmdUoR3
This commit is contained in:
parent
44526254d0
commit
fa1fa5aafd
5 changed files with 526 additions and 4 deletions
224
tests/test_mandate.py
Normal file
224
tests/test_mandate.py
Normal file
|
|
@ -0,0 +1,224 @@
|
|||
"""Unit tests for the run MANDATE IR + its fail-fast loader (Trekk A1).
|
||||
|
||||
The mandate is what a domain expert commissions a run with: what shall be evaluated
|
||||
(named approaches, and/or the system's own proposals) and what the run is meant to
|
||||
achieve (``objective`` / ``success_criteria``). It is *authoritative startup config* —
|
||||
so loading mirrors ``dimension.load_dimension`` / ``contracts.load_goal_config``
|
||||
(missing -> ``FileNotFoundError``, malformed -> ``ValidationError``), NOT the tolerant
|
||||
RAW verdict-inbox layer.
|
||||
|
||||
Two refusals carry real defect classes and are tested explicitly:
|
||||
|
||||
* an EMPTY commission (no approaches AND no own proposals) — a run with nothing to do
|
||||
is a caller error, not a result (mirrors ``BudgetRefused``'s startup refusal);
|
||||
* a DUPLICATE approach id — ``id`` is the coverage-report key, so two rows sharing one
|
||||
key would silently collapse into one (the S3.2 key-collision class), and the reserved
|
||||
``OWN_PROPOSAL_ID`` would collide with the system's own row the same way.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
|
||||
from portfolio_optimiser.mandate import (
|
||||
OWN_PROPOSAL_ID,
|
||||
Approach,
|
||||
Mandate,
|
||||
announce,
|
||||
load_mandate,
|
||||
)
|
||||
|
||||
|
||||
def _mandate(**overrides: object) -> Mandate:
|
||||
kwargs: dict[str, object] = {
|
||||
"objective": "Find operating measures that can be executed in 2026 without rebuilding.",
|
||||
"approaches": (
|
||||
Approach(id="led-retrofit", label="LED retrofit of office lighting"),
|
||||
Approach(id="service-contract", label="Renegotiate the service contract"),
|
||||
),
|
||||
}
|
||||
kwargs.update(overrides)
|
||||
return Mandate(**kwargs) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_mandate_defaults_allow_own_proposals() -> None:
|
||||
"""``allow_own_proposals`` defaults to True: naming approaches never silently forbids
|
||||
the system from adding its own (the operator's 'og/eller')."""
|
||||
assert _mandate().allow_own_proposals is True
|
||||
|
||||
|
||||
def test_mandate_accepts_named_approaches_only() -> None:
|
||||
"""'Evaluate ONLY these' is a legal commission."""
|
||||
m = _mandate(allow_own_proposals=False)
|
||||
assert [a.id for a in m.approaches] == ["led-retrofit", "service-contract"]
|
||||
assert m.allow_own_proposals is False
|
||||
|
||||
|
||||
def test_mandate_accepts_own_proposals_only() -> None:
|
||||
"""'Find your own' with no named approaches is a legal commission (today's behaviour,
|
||||
now stated rather than implied)."""
|
||||
m = _mandate(approaches=())
|
||||
assert m.approaches == ()
|
||||
assert m.allow_own_proposals is True
|
||||
|
||||
|
||||
def test_mandate_empty_commission_refused() -> None:
|
||||
"""No approaches AND no own proposals = a run with nothing to do -> refused at
|
||||
construction, never started."""
|
||||
with pytest.raises(ValidationError):
|
||||
_mandate(approaches=(), allow_own_proposals=False)
|
||||
|
||||
|
||||
def test_mandate_duplicate_approach_id_refused() -> None:
|
||||
"""``id`` keys the coverage report; two rows on one key would collapse silently."""
|
||||
with pytest.raises(ValidationError):
|
||||
_mandate(
|
||||
approaches=(
|
||||
Approach(id="led-retrofit", label="LED retrofit"),
|
||||
Approach(id="led-retrofit", label="LED retrofit, second take"),
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def test_mandate_reserved_own_proposal_id_refused() -> None:
|
||||
"""The system's own row is reported under ``OWN_PROPOSAL_ID``; an expert approach
|
||||
claiming that id would collide with it."""
|
||||
with pytest.raises(ValidationError):
|
||||
_mandate(approaches=(Approach(id=OWN_PROPOSAL_ID, label="Sneaky"),))
|
||||
|
||||
|
||||
def test_mandate_requires_an_objective() -> None:
|
||||
"""An empty objective defeats the whole point of the mandate (krav 2): a run must say
|
||||
what it is for."""
|
||||
with pytest.raises(ValidationError):
|
||||
_mandate(objective="")
|
||||
|
||||
|
||||
def test_approach_requires_id_and_label() -> None:
|
||||
"""Both the coverage key and its human label must be non-empty."""
|
||||
with pytest.raises(ValidationError):
|
||||
Approach(id="", label="No id")
|
||||
with pytest.raises(ValidationError):
|
||||
Approach(id="no-label", label="")
|
||||
|
||||
|
||||
# --- fail-fast loader (mirrors load_dimension / load_goal_config) --------------------------------
|
||||
|
||||
|
||||
def test_load_mandate_round_trip(tmp_path) -> None:
|
||||
"""A valid mandate JSON round-trips through ``load_mandate`` (accepts str | Path)."""
|
||||
m = _mandate()
|
||||
p = tmp_path / "mandate.json"
|
||||
p.write_text(m.model_dump_json(), encoding="utf-8")
|
||||
assert load_mandate(p) == m
|
||||
assert load_mandate(str(p)) == m
|
||||
|
||||
|
||||
def test_load_mandate_missing_file_raises(tmp_path) -> None:
|
||||
"""Missing file fails fast — authoritative startup config, not a tolerant RAW inbox."""
|
||||
with pytest.raises(FileNotFoundError):
|
||||
load_mandate(tmp_path / "does-not-exist.json")
|
||||
|
||||
|
||||
def test_load_mandate_malformed_shape_raises(tmp_path) -> None:
|
||||
"""Malformed content (missing ``objective``) fails fast with ``ValidationError``."""
|
||||
bad = tmp_path / "mandate.json"
|
||||
bad.write_text('{"approaches": []}', encoding="utf-8")
|
||||
with pytest.raises(ValidationError):
|
||||
load_mandate(bad)
|
||||
|
||||
|
||||
def test_load_mandate_not_json_raises(tmp_path) -> None:
|
||||
"""Non-JSON content fails fast too (never read as 'no mandate')."""
|
||||
bad = tmp_path / "mandate.json"
|
||||
bad.write_text("this is not json", encoding="utf-8")
|
||||
with pytest.raises(ValidationError):
|
||||
load_mandate(bad)
|
||||
|
||||
|
||||
# --- the run announcement (Trekk A2): what this run will do, before the first paid call ----------
|
||||
|
||||
|
||||
def _announce(m: Mandate | None = None, **overrides: object) -> str:
|
||||
kwargs: dict[str, object] = {
|
||||
"project_id": "BYGG-KONTOR-NORD",
|
||||
"max_rounds": 3,
|
||||
"max_tokens": 100_000,
|
||||
}
|
||||
kwargs.update(overrides)
|
||||
return announce(m if m is not None else _mandate(), **kwargs) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_announce_names_every_commissioned_approach() -> None:
|
||||
"""Every approach the expert commissioned is named BEFORE the run starts — the operator must
|
||||
be able to see what was ordered without reading the config file back."""
|
||||
text = _announce()
|
||||
assert "led-retrofit" in text
|
||||
assert "service-contract" in text
|
||||
assert "LED retrofit of office lighting" in text
|
||||
|
||||
|
||||
def test_announce_states_the_objective_and_success_criteria() -> None:
|
||||
"""Krav 2: the run says what it is for, and what would count as success."""
|
||||
m = _mandate(success_criteria="At least one measure that passes the validator.")
|
||||
text = _announce(m)
|
||||
assert "without rebuilding" in text
|
||||
assert "At least one measure that passes the validator." in text
|
||||
|
||||
|
||||
def test_announce_distinguishes_only_these_from_these_plus_own() -> None:
|
||||
"""The 'og/eller' choice is visible in the announcement, not buried in the config."""
|
||||
both = _announce(_mandate(allow_own_proposals=True))
|
||||
only = _announce(_mandate(allow_own_proposals=False))
|
||||
assert "own proposals" in both
|
||||
assert "only these" in only
|
||||
assert both != only
|
||||
|
||||
|
||||
def test_announce_without_approaches_says_so() -> None:
|
||||
"""A 'find your own' commission announces that too — never a blank line where the list was."""
|
||||
text = _announce(_mandate(approaches=()))
|
||||
assert "own proposals" in text
|
||||
|
||||
|
||||
def test_announce_omits_optional_lines_when_absent() -> None:
|
||||
"""Dimension and target lines appear ONLY when the run actually has them — an announcement
|
||||
must not imply a scope or a target that was never configured."""
|
||||
text = _announce()
|
||||
assert "Scoped to" not in text
|
||||
assert "Target" not in text
|
||||
|
||||
|
||||
def test_announce_includes_scope_and_target_when_given() -> None:
|
||||
"""...and they ARE stated when configured (the target is restated from the goal config, which
|
||||
remains its one home)."""
|
||||
text = _announce(dimension_label="energi (cost codes ENERGI-*)", goal_nok=150_000.0)
|
||||
assert "energi (cost codes ENERGI-*)" in text
|
||||
assert "150000" in text
|
||||
|
||||
|
||||
def test_announce_declares_no_egress_by_default() -> None:
|
||||
"""The egress declaration is present and explicit even when nothing is contacted — silence
|
||||
would read the same as 'not checked' (repo invariant: no silent egress)."""
|
||||
assert "no external services" in _announce()
|
||||
|
||||
|
||||
def test_announce_names_every_external_service() -> None:
|
||||
"""Trekk B's egress declaration: every server that may be contacted is named up front."""
|
||||
text = _announce(external_services=("prisregister", "maalerdata"))
|
||||
assert "prisregister" in text
|
||||
assert "maalerdata" in text
|
||||
assert "no external services" not in text
|
||||
|
||||
|
||||
def test_announce_states_the_caps() -> None:
|
||||
"""Stop criteria + budget cap are part of 'what this run will do' (fail-fast invariant)."""
|
||||
text = _announce()
|
||||
assert "3 rounds" in text
|
||||
assert "100000 tokens" in text
|
||||
|
||||
|
||||
def test_announce_is_deterministic() -> None:
|
||||
"""Byte-stable: no wall clock, no set-iteration ordering — so it can be golden-tested."""
|
||||
assert _announce() == _announce()
|
||||
100
tests/test_mandate_generation_loadbearing.py
Normal file
100
tests/test_mandate_generation_loadbearing.py
Normal file
|
|
@ -0,0 +1,100 @@
|
|||
"""Load-bearing: a COMMISSIONED approach reaches the proposer, and gets no discount from the
|
||||
deterministic validator (Trekk A3).
|
||||
|
||||
The seam this pins is the whole point of krav 1: a domain expert names an approach, and the
|
||||
hypothesis prompt must actually carry it — otherwise the run merely *claims* to evaluate what was
|
||||
ordered. Two detach points, and each must go RED on its own:
|
||||
|
||||
* drop the approach block from ``_build_messages`` -> the expert's label/description never reaches
|
||||
the model (``test_commissioned_approach_reaches_the_proposer``);
|
||||
* let a commissioned approach bypass ``validate_proposal`` -> an expert's wish would outrank the
|
||||
deterministic gate (``test_commissioned_approach_gets_no_validator_discount``).
|
||||
|
||||
The control (``test_no_approach_keeps_the_base_prompt_byte_identical``) proves the addition is
|
||||
inert when no mandate is given: a test that can only go green proves nothing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from spikes._harness import FakeChatClient, message_texts
|
||||
|
||||
from portfolio_optimiser.budget import Budget, TokenMeter
|
||||
from portfolio_optimiser.generate import _build_messages, generate_via_llm
|
||||
from portfolio_optimiser.mandate import Approach
|
||||
from portfolio_optimiser.reference_domain import load_reference_projects
|
||||
from portfolio_optimiser.validator import Rejection, ValidatedProposal
|
||||
|
||||
_VALID = (
|
||||
'{"project_id":"FV42-GSV-E1","measure":"Reduce scope",'
|
||||
'"affected_items":[{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"03.1","quantity":1800,"unit_cost":310}],"claimed_saving_nok":200000}'
|
||||
)
|
||||
#: Same shape and a value the IR itself accepts, but above the method cap (30% of the ~1.48M
|
||||
#: affected total = ~445k) -> the DETERMINISTIC VALIDATOR is what must reject it, not the parser.
|
||||
_INFEASIBLE = (
|
||||
'{"project_id":"FV42-GSV-E1","measure":"Reduce scope",'
|
||||
'"affected_items":[{"code":"05.2","quantity":4300,"unit_cost":215},'
|
||||
'{"code":"03.1","quantity":1800,"unit_cost":310}],"claimed_saving_nok":800000}'
|
||||
)
|
||||
|
||||
_APPROACH = Approach(
|
||||
id="led-retrofit",
|
||||
label="LED retrofit of office lighting",
|
||||
description="Operations believes the fixtures are original and run far past their rated life.",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def project():
|
||||
return load_reference_projects()[0] # FV42-GSV-E1
|
||||
|
||||
|
||||
def _meter() -> TokenMeter:
|
||||
return TokenMeter(Budget(max_tokens=10**9, max_rounds=20))
|
||||
|
||||
|
||||
def test_commissioned_approach_reaches_the_proposer(project) -> None:
|
||||
"""The expert's own words — label AND description — reach the hypothesis prompt VERBATIM.
|
||||
|
||||
The description is the part the model cannot infer from cost data (it is the expert's reason
|
||||
for wanting this tried), so paraphrasing it away would quietly discard the domain knowledge.
|
||||
"""
|
||||
text = message_texts(_build_messages(project, "ctx", approach=_APPROACH))[0]
|
||||
assert _APPROACH.label in text
|
||||
assert _APPROACH.description in text
|
||||
|
||||
|
||||
def test_no_approach_keeps_the_base_prompt_byte_identical(project) -> None:
|
||||
"""CONTROL: without a mandate the prompt is byte-identical to the pre-Trekk-A one, so every
|
||||
existing run and golden is untouched (mirrors ``prior_rejection=None``)."""
|
||||
base = message_texts(_build_messages(project, "ctx"))[0]
|
||||
assert base == message_texts(_build_messages(project, "ctx", approach=None))[0]
|
||||
assert "Propose ONE concrete cost-saving measure for this project." in base
|
||||
assert _APPROACH.label not in base
|
||||
|
||||
|
||||
async def test_commissioned_approach_gets_no_validator_discount(project) -> None:
|
||||
"""A commissioned approach whose numbers do not hold is REJECTED — the deterministic gate is
|
||||
blocking, and being asked for by a domain expert is not a reason to pass it.
|
||||
|
||||
This is the one rule krav 1 must not be allowed to erode: the expert directs *what is
|
||||
evaluated*, never *what is approved*.
|
||||
"""
|
||||
client = FakeChatClient(scripted=[_INFEASIBLE], default_reply=_INFEASIBLE)
|
||||
result = await generate_via_llm(
|
||||
client, project, "", _meter(), max_attempts=1, approach=_APPROACH
|
||||
)
|
||||
assert isinstance(result, Rejection)
|
||||
|
||||
|
||||
async def test_commissioned_approach_still_validates_when_the_numbers_hold(project) -> None:
|
||||
"""...and the same commissioned path DOES produce a validated proposal when the numbers are
|
||||
feasible — so the rejection above is the validator working, not the approach path being
|
||||
broken end to end."""
|
||||
client = FakeChatClient(scripted=[_VALID], default_reply=_VALID)
|
||||
result = await generate_via_llm(
|
||||
client, project, "", _meter(), max_attempts=1, approach=_APPROACH
|
||||
)
|
||||
assert isinstance(result, ValidatedProposal)
|
||||
assert _APPROACH.label in client.received_texts[0][0] # it went through the commissioned prompt
|
||||
|
|
@ -22,6 +22,7 @@ from portfolio_optimiser import okf
|
|||
_MAF_FREE_MODULES = [
|
||||
"okf.py",
|
||||
"dimension.py",
|
||||
"mandate.py",
|
||||
"outbox.py",
|
||||
"costsim.py",
|
||||
"hitl.py",
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue