P18's round 2 ended with two VALIDATED proposals whose affected_item codes were ordinary words from a road standard's prose -- impulsventilator (4 of 270 N500 documents) and bituminoest baerelag (4 of 1133 N200). Both are grounded in P7's sense and neither is inert in P18/B1's sense; they are simply not identifiers of a cost line, and the gate had no stage that could say so. Known positive MEASURED, not asserted: replayed offline against the bases those runs were given, both come back Rejection naming the denominator. IDENTIFIER_FORMS moved from generate.py to validator.py: they now drive both P8's report and this gate, and two copies of "what an identifier looks like" would let the two disagree about one run's own input. B1 -- two new forms, transcribed from measurement. R761's requirement numbers are bare dotted numbers and all six refs in kontrakt-sorasen's fasit are of that shape, which neither pre-P19 form matched: r761's whole offer was 3 identifiers over 6.5 MB, and is now 2332. The FIRST form was widened in the same pass because B2 made these forms decide prose vs identifier, and this repo's own ENERGI-TOTAL-EL matched none of them -- a gate may only be wrong in the direction that admits too much. Three things keep the gate from being a rule about shapes: the generality guard (it fires only where the input offers forms), the baseline exemption (stage 0 has already ruled that code real), and full-matching. Honesty limit, measured and given its OWN arm: a decimal and an R761 process number are typographically identical, so the form counts both. P8's existing "bare numbers" arm is narrowed to bare INTEGERS accordingly. Measured over all nine round-1+2 outboxes: 26 of 36 codes are prose. Load-bearing measured (22 arms), four mutations all red against the whole suite (16 / 10 / 1 / 2), green control 1734/5 and the golden byte-unchanged. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
255 lines
10 KiB
Python
255 lines
10 KiB
Python
"""S2.1 outbox output-layer (målbilde §3, R2 RAW layer): ``write_outbox`` persists two
|
|
byte-deterministic JSON artefacts per run — ``{run_id}-proposal.json`` (the candidate + provenance)
|
|
and ``{run_id}-outcome.json`` (the validated percentiles OR the rejection reason, plus the checker
|
|
verdict + verdict id). This is the traceable output layer Fase 5 (S5.1/S5.4) builds on.
|
|
|
|
The writer is MAF-free (pure stdlib + the MAF-free ``validator`` leaf) — proven load-bearing by
|
|
``test_okf_is_maf_free`` covering ``outbox.py`` (T-2.1d), plus the byte-determinism (T-2.1b) and
|
|
both-arms (T-2.1a) checks below.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Callable
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
from agent_framework import BaseChatClient
|
|
from conftest import SyntheticUsageChatClient
|
|
|
|
from portfolio_optimiser.ir import AffectedItem, SavingsProposal
|
|
from portfolio_optimiser.outbox import write_outbox, write_run_config
|
|
from portfolio_optimiser.provenance import Citation, ProvenanceStamp
|
|
from portfolio_optimiser.retrieval import TextSpan
|
|
from portfolio_optimiser.run import run_project
|
|
from portfolio_optimiser.validator import Rejection, ValidatedProposal
|
|
|
|
# --- Step 6 (run_project wiring) fixtures: the shared bundle + a role-aware scripted factory ---
|
|
BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
|
|
|
|
# A VALIDATOR-VALID BYGG-KONTOR-NORD proposal (degenerate Monte Carlo P90 = 0.30 x 300000 = 90000
|
|
# >= claimed 30000 -> validates), so the only possible rejecter is the checker.
|
|
_VALID_PROPOSER_REPLY = (
|
|
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
|
|
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
|
|
)
|
|
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"}
|
|
|
|
|
|
def _role_factory(proposer_reply: str, checker_reply: str) -> Callable[[str], BaseChatClient]:
|
|
"""A role-aware synthetic factory: the checker speaks its verdict, the proposer its proposal."""
|
|
|
|
def factory(role: str) -> BaseChatClient:
|
|
return SyntheticUsageChatClient(
|
|
default_reply=checker_reply if role == "checker" else proposer_reply
|
|
)
|
|
|
|
return factory
|
|
|
|
|
|
_PROPOSAL = SavingsProposal(
|
|
project_id="P1",
|
|
measure="LED-retrofit av kontorbelysning",
|
|
affected_items=[AffectedItem(code="ENERGI-TOTAL-EL", quantity=1000.0, unit_cost=1.0)],
|
|
claimed_saving_nok=200.0,
|
|
assumptions={"ENERGI-TOTAL-EL": (0.8, 1.2)},
|
|
)
|
|
_PROVENANCE = ProvenanceStamp(
|
|
citations=[
|
|
Citation(file="f.md", locator=TextSpan(start_index=0, end_index=5), snippet="hello")
|
|
],
|
|
model="synthetic",
|
|
role="proposer",
|
|
validator_decision="validated",
|
|
token_usage=8,
|
|
# These fixtures stand in for an ordinary complete run; the road path is anchored by
|
|
# construction, so ``True`` is the honest value here. The un-anchored case has its own file.
|
|
cost_baseline_anchored=True,
|
|
bundle_id_source=None,
|
|
code_forms={},
|
|
)
|
|
_VALIDATED = ValidatedProposal(
|
|
proposal=_PROPOSAL, p10=100.0, p50=150.0, p90=200.0, nominal_feasible=180.0
|
|
)
|
|
_REJECTION = Rejection(proposal=_PROPOSAL, reason="checker rejected: unsupported reasoning")
|
|
|
|
|
|
def _read(path: Path) -> str:
|
|
return path.read_text(encoding="utf-8")
|
|
|
|
|
|
def test_outbox_writes_validated_arm(tmp_path) -> None:
|
|
"""T-2.1a (validated arm): both files are written with the proposal fields + provenance in
|
|
proposal.json, and outcome_type=validated + percentiles + checker verdict + verdict id in
|
|
outcome.json."""
|
|
prop_path, out_path = write_outbox(
|
|
str(tmp_path),
|
|
"run-1",
|
|
outcome=_VALIDATED,
|
|
provenance=_PROVENANCE,
|
|
checker_verdict="approve",
|
|
verdict_id="vid-abc",
|
|
)
|
|
assert prop_path == tmp_path / "run-1-proposal.json"
|
|
assert out_path == tmp_path / "run-1-outcome.json"
|
|
|
|
proposal_text = _read(prop_path)
|
|
assert "LED-retrofit av kontorbelysning" in proposal_text
|
|
assert "ENERGI-TOTAL-EL" in proposal_text
|
|
assert '"model": "synthetic"' in proposal_text # provenance embedded
|
|
|
|
outcome_text = _read(out_path)
|
|
assert '"outcome_type": "validated"' in outcome_text
|
|
assert '"p90": 200.0' in outcome_text
|
|
assert '"checker_verdict": "approve"' in outcome_text
|
|
assert '"verdict_id": "vid-abc"' in outcome_text
|
|
assert "reason" not in outcome_text # the validated arm carries no rejection reason
|
|
|
|
|
|
def test_outbox_writes_rejected_arm(tmp_path) -> None:
|
|
"""T-2.1a (rejected arm): a Rejection outcome writes outcome_type=rejected + the reason, and NO
|
|
percentiles (the Rejection type carries none — it can never masquerade as validated)."""
|
|
_prop_path, out_path = write_outbox(
|
|
str(tmp_path),
|
|
"run-2",
|
|
outcome=_REJECTION,
|
|
provenance=_PROVENANCE,
|
|
checker_verdict="reject",
|
|
verdict_id="vid-def",
|
|
)
|
|
outcome_text = _read(out_path)
|
|
assert '"outcome_type": "rejected"' in outcome_text
|
|
assert "unsupported reasoning" in outcome_text
|
|
assert '"verdict_id": "vid-def"' in outcome_text
|
|
assert "p90" not in outcome_text # no percentiles on the rejected arm
|
|
|
|
|
|
def test_outbox_is_byte_deterministic(tmp_path) -> None:
|
|
"""T-2.1b: two writes with identical input + the same run_id produce byte-identical files
|
|
(sort_keys + indent=2 + explicit LF, no wall-clock/uuid) — the artefacts are diff-stable."""
|
|
a = tmp_path / "a"
|
|
b = tmp_path / "b"
|
|
args = dict(
|
|
outcome=_VALIDATED, provenance=_PROVENANCE, checker_verdict="approve", verdict_id="vid-abc"
|
|
)
|
|
pa, oa = write_outbox(str(a), "run-1", **args)
|
|
pb, ob = write_outbox(str(b), "run-1", **args)
|
|
assert pa.read_bytes() == pb.read_bytes()
|
|
assert oa.read_bytes() == ob.read_bytes()
|
|
|
|
|
|
def test_write_run_config_byte_deterministic_and_fields(tmp_path) -> None:
|
|
"""S4.2 T-4.2a: ``write_run_config`` writes ``{run_id}-runconfig.json`` carrying profile +
|
|
resolved model per BUILT role + params + token cap; two identical writes are byte-identical
|
|
(sort_keys/indent/LF, no wall-clock), and the payload leaks NO date/timestamp key (§4 pkt 3
|
|
excludes wall-clock for byte-determinism). Drop a field, break ``_dump`` determinism, or leak a
|
|
timestamp → RED."""
|
|
import json
|
|
|
|
a = tmp_path / "a"
|
|
b = tmp_path / "b"
|
|
kwargs = dict(
|
|
profile="local",
|
|
resolved_models={"proposer": "qwen3:4b", "checker": "qwen3:4b"},
|
|
max_rounds=3,
|
|
max_tokens=100_000,
|
|
top_k=3,
|
|
)
|
|
pa = write_run_config(str(a), "run-1", **kwargs)
|
|
pb = write_run_config(str(b), "run-1", **kwargs)
|
|
assert pa.name == "run-1-runconfig.json"
|
|
assert pa.read_bytes() == pb.read_bytes() # byte-deterministic
|
|
text = pa.read_text(encoding="utf-8")
|
|
assert text.endswith("\n")
|
|
payload = json.loads(text)
|
|
assert payload["run_id"] == "run-1"
|
|
assert payload["profile"] == "local"
|
|
assert payload["resolved_models"] == {"proposer": "qwen3:4b", "checker": "qwen3:4b"}
|
|
assert payload["max_rounds"] == 3
|
|
assert payload["max_tokens"] == 100_000
|
|
assert payload["top_k"] == 3
|
|
assert not any(k in payload for k in ("date", "timestamp", "created", "ts"))
|
|
|
|
|
|
def test_outbox_registered_maf_free() -> None:
|
|
"""T-2.1d meta: outbox.py is registered in the MAF-free guard list, so test_okf_is_maf_free
|
|
actually scans it — otherwise the MAF-free claim would be green-but-dead (never checked)."""
|
|
from tests.test_okf import _MAF_FREE_MODULES
|
|
|
|
assert "outbox.py" in _MAF_FREE_MODULES
|
|
|
|
|
|
async def test_run_project_writes_outbox_validated_arm(tmp_path) -> None:
|
|
"""T-2.1a (via run_project, validated): a run with an ``outbox_dir`` + ``run_id`` writes both
|
|
artefacts; the outcome file records ``validated``. Detach the ``write_outbox`` call in
|
|
``run_project`` → the files are absent → RED."""
|
|
factory = _role_factory(_VALID_PROPOSER_REPLY, "VERDICT: APPROVE")
|
|
result = await run_project(
|
|
"BYGG-KONTOR-NORD",
|
|
"local",
|
|
docs_dir=str(BUNDLE_DIR),
|
|
bundle_dir=str(BUNDLE_DIR),
|
|
verdict_input=_VERDICT_INPUT,
|
|
client_factory=factory,
|
|
outbox_dir=str(tmp_path),
|
|
run_id="run-validated",
|
|
)
|
|
assert isinstance(result.outcome, ValidatedProposal)
|
|
proposal_file = tmp_path / "run-validated-proposal.json"
|
|
outcome_file = tmp_path / "run-validated-outcome.json"
|
|
assert proposal_file.is_file()
|
|
assert outcome_file.is_file()
|
|
assert '"outcome_type": "validated"' in outcome_file.read_text(encoding="utf-8")
|
|
|
|
|
|
async def test_run_project_writes_outbox_rejected_arm(tmp_path) -> None:
|
|
"""T-2.1a (via run_project, rejected): a checker ``VERDICT: REJECT`` flips an otherwise-validated
|
|
proposal to a Rejection, and the outbox outcome file records ``rejected``."""
|
|
factory = _role_factory(
|
|
_VALID_PROPOSER_REPLY, "VERDICT: REJECT - payback exceeds horizon (sim)"
|
|
)
|
|
result = await run_project(
|
|
"BYGG-KONTOR-NORD",
|
|
"local",
|
|
docs_dir=str(BUNDLE_DIR),
|
|
bundle_dir=str(BUNDLE_DIR),
|
|
verdict_input=_VERDICT_INPUT,
|
|
client_factory=factory,
|
|
outbox_dir=str(tmp_path),
|
|
run_id="run-rejected",
|
|
)
|
|
assert isinstance(result.outcome, Rejection)
|
|
outcome_file = tmp_path / "run-rejected-outcome.json"
|
|
assert '"outcome_type": "rejected"' in outcome_file.read_text(encoding="utf-8")
|
|
|
|
|
|
async def test_run_project_without_outbox_dir_writes_nothing(tmp_path) -> None:
|
|
"""T-2.1c (control): a run WITHOUT ``outbox_dir`` writes no artefacts — proving the writes above
|
|
are caused by the seam, not incidental."""
|
|
factory = _role_factory(_VALID_PROPOSER_REPLY, "VERDICT: APPROVE")
|
|
await run_project(
|
|
"BYGG-KONTOR-NORD",
|
|
"local",
|
|
docs_dir=str(BUNDLE_DIR),
|
|
bundle_dir=str(BUNDLE_DIR),
|
|
verdict_input=_VERDICT_INPUT,
|
|
client_factory=factory,
|
|
)
|
|
assert list(tmp_path.iterdir()) == []
|
|
|
|
|
|
async def test_run_project_outbox_dir_requires_run_id(tmp_path) -> None:
|
|
"""T-2.1e: an ``outbox_dir`` with ``run_id=None`` raises (no wall-clock/uuid default — the
|
|
artefacts are byte-deterministic and keyed on run_id)."""
|
|
factory = _role_factory(_VALID_PROPOSER_REPLY, "VERDICT: APPROVE")
|
|
with pytest.raises(ValueError):
|
|
await run_project(
|
|
"BYGG-KONTOR-NORD",
|
|
"local",
|
|
docs_dir=str(BUNDLE_DIR),
|
|
bundle_dir=str(BUNDLE_DIR),
|
|
verdict_input=_VERDICT_INPUT,
|
|
client_factory=factory,
|
|
outbox_dir=str(tmp_path),
|
|
run_id=None,
|
|
)
|