python -m portfolio_optimiser.evals.round_builder --outbox <dir> --round <n> --ran-at <ISO> writes <rounds-dir>/<n>/ with the run's artefacts COPIED in, outcome.json derived from that copy, and report.md -- the one artefact in a round a domain expert reads and corrects. Round 0 of the v1 criterion can now be made; it counted 0 of 3 because it could not be, which is a different failure from a round nobody had held. What it derives it derives with the gate's own functions rather than a second copy: verify_run decides whether the run stands up to itself (an artefact contradicting its coverage row, a half-missing family and a stray artefact are all refused AT THE SOURCE, before a byte is written), stage_of gives column (c), row_changed gives the report's "changed since the previous round", parse_time refuses a stamp without a zone, safe_rounds_dir refuses a round directory the repo would commit. The validated total is ledger.to_ore per amount, summed as integers. Two things it never does, and both are the point. It never writes the operator's attestation -- the gate stops at FORM OK without one, and that is correct, because no arrangement of files can witness that a run happened. And it never invents: --ran-at is required because no outbox artefact carries a clock, and feedback_ids stays empty because no run records which feedback item produced which row. The report says "ingen tilbakemelding forklarer dette" on every changed row rather than hiding that model noise and an answered objection look alike. Chosen and why: --ran-at as a required argument rather than the coverage file's mtime, because an mtime is a filesystem attribute one call sets and reading it as evidence made row 2 green on a tree nothing had run in (18.09). The report carries no raw stage identifier -- every stage sentence is "<short name>: <explanation>" so the one-line diff of what changed has words a reader can act on. A citation shows its COUNT, because a run that cited 446 places and one that cited one must not look the same. [skip-docs]: the ledger row is in docs/invarianter.md, which is where this repo's rules live. README is the product's front door and this is an operator tool behind `python -m`, the same class as costsim/hitl/preflight, which README deliberately does not carry; v1-rounds/ is gitignored internal machinery and the gate itself is not in README either. CLAUDE.md was emptied of exactly this kind of row in session 130 and is not the place to put one back. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
716 lines
31 KiB
Python
716 lines
31 KiB
Python
"""The round builder's own tests: what it derives, what it refuses, and what it must never write.
|
|
|
|
Every number asserted here is counted a SECOND time from ``_SPEC`` — the fixture's own table —
|
|
rather than read back from the builder. A test that asks the builder what it produced and then
|
|
agrees with it cannot tell two implementations apart; the table is the independent denominator.
|
|
|
|
The fixture outbox is written by the PRODUCT (``outbox.write_outbox`` / ``write_coverage``), for
|
|
``test_v1_gate._real_run_family``'s reason: if the builder is measured against a shape this file
|
|
invents, it is measured against a fixture. The end of the chain is checked the same way — the
|
|
built round is handed to ``v1_gate`` itself, which must read it as FORM OK rather than as a round
|
|
it cannot parse.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from portfolio_optimiser.evals import round_builder as rb
|
|
from portfolio_optimiser.evals import v1_gate as gate
|
|
from portfolio_optimiser.ir import AffectedItem, SavingsProposal
|
|
from portfolio_optimiser.ledger import to_ore
|
|
from portfolio_optimiser.outbox import write_coverage, write_outbox
|
|
from portfolio_optimiser.provenance import Citation, ProvenanceStamp
|
|
from portfolio_optimiser.retrieval import TextSpan
|
|
from portfolio_optimiser.validator import (
|
|
UNSUPPORTED_REASON,
|
|
Rejection,
|
|
Unsupported,
|
|
ValidatedProposal,
|
|
)
|
|
from portfolio_optimiser.verdicts import features_from_ir, verdict_key
|
|
|
|
_RUN = "demo-kommune-2027-01"
|
|
#: A run happens at a wall-clock time no artefact records, so the operator states it. Fixed here.
|
|
_RAN_AT = "2026-09-18T11:20:00+02:00"
|
|
_ATTEST_FILE = "attestering.txt"
|
|
|
|
#: The fixture run, as a table. Columns: id, label, coverage status, validated NOK, the claim the
|
|
#: proposal itself carried, the detail the coverage row carries. Everything this file asserts is
|
|
#: counted from HERE, never from the builder's own answer.
|
|
#:
|
|
#: The two 60000.005 amounts are deliberate: quantizing per amount and summing the integers
|
|
#: (kø-(p), ``ledger.to_ore``) gives 137000002 øre, while summing the floats first gives
|
|
#: 137000001. One øre is the whole distance between the rule and its most plausible violation.
|
|
_SPEC: tuple[tuple[str, str, str, float | None, float, str], ...] = (
|
|
(
|
|
"a1-kortere-rekkverk",
|
|
"Kortere rekkverk langs fv. 12",
|
|
"validated",
|
|
1_250_000.0,
|
|
1_250_000.0,
|
|
"",
|
|
),
|
|
(
|
|
"a2-faerre-kummer",
|
|
"Færre kummer i kryssene",
|
|
"rejected",
|
|
None,
|
|
900_000.0,
|
|
"claimed saving 900000 exceeds P90 feasible 410000",
|
|
),
|
|
(
|
|
"a3-tynnere-dekke",
|
|
"Tynnere dekke på gang- og sykkelvegen",
|
|
"validated",
|
|
60_000.005,
|
|
60_000.005,
|
|
"",
|
|
),
|
|
(
|
|
"a4-smalere-skulder",
|
|
"Smalere skulder på strekningen",
|
|
"unsupported",
|
|
None,
|
|
400_000.0,
|
|
UNSUPPORTED_REASON,
|
|
),
|
|
(
|
|
"a5-enklere-rekkverksender",
|
|
"Enklere rekkverksender",
|
|
"validated",
|
|
60_000.005,
|
|
60_000.005,
|
|
"",
|
|
),
|
|
(
|
|
"a6-enklere-belysning",
|
|
"Enklere belysning i krysset",
|
|
"not_evaluated",
|
|
None,
|
|
0.0,
|
|
"tokens",
|
|
),
|
|
)
|
|
|
|
_COMMISSIONED = len(_SPEC)
|
|
_EVALUATED = tuple(aid for aid, _, status, *_ in _SPEC if status != "not_evaluated")
|
|
_NOT_EVALUATED = tuple(aid for aid, _, status, *_ in _SPEC if status == "not_evaluated")
|
|
_VALIDATED = tuple(aid for aid, _, status, *_ in _SPEC if status == "validated")
|
|
_FELL = tuple(aid for aid, _, status, *_ in _SPEC if status in ("rejected", "unsupported"))
|
|
#: Counted from the table, per amount, with the framework's one conversion (kø-(p)).
|
|
_VALIDATED_ORE = sum(to_ore(nok) for _, _, status, nok, _, _ in _SPEC if status == "validated")
|
|
_STOP_REASON = "tokens"
|
|
|
|
|
|
def _measure(aid: str) -> str:
|
|
return f"Tiltak: {dict((a, label) for a, label, *_ in _SPEC)[aid]}"
|
|
|
|
|
|
def _ir(aid: str, claimed: float) -> dict[str, Any]:
|
|
"""The proposal IR a run would have written — built through the real model, so the fixture
|
|
cannot drift from the shape ``write_outbox`` persists and the gate re-mints the key from."""
|
|
return SavingsProposal(
|
|
project_id="demo-kommune",
|
|
measure=_measure(aid),
|
|
affected_items=[
|
|
AffectedItem(code=f"DK-{aid[:2].upper()}", quantity=1.0, unit_cost=claimed)
|
|
],
|
|
claimed_saving_nok=claimed,
|
|
).model_dump()
|
|
|
|
|
|
def _stamp(decision: str) -> ProvenanceStamp:
|
|
return ProvenanceStamp(
|
|
citations=[
|
|
Citation(
|
|
file="krav/N100/id-1ba48872.md",
|
|
locator=TextSpan(start_index=0, end_index=48),
|
|
snippet="Paa motorveger anbefales restriktiv bruk av kryss.",
|
|
)
|
|
],
|
|
model="synthetic",
|
|
role="proposer",
|
|
validator_decision=decision, # type: ignore[arg-type]
|
|
token_usage=8,
|
|
cost_baseline_anchored=True,
|
|
bundle_id_source=None,
|
|
code_forms={},
|
|
)
|
|
|
|
|
|
def _coverage_rows(spec: Any = _SPEC) -> list[dict[str, Any]]:
|
|
return [
|
|
{"id": aid, "label": label, "status": status, "detail": detail, "saving_nok": nok}
|
|
for aid, label, status, nok, _claimed, detail in spec
|
|
]
|
|
|
|
|
|
def _outbox(root: Path, run_id: str = _RUN, spec: Any = _SPEC) -> Path:
|
|
"""One real run's outbox: a proposal/outcome pair per EVALUATED approach plus the coverage."""
|
|
outbox = root / "kjoring"
|
|
for aid, _label, status, nok, claimed, detail in spec:
|
|
if status == "not_evaluated":
|
|
continue
|
|
ir = _ir(aid, claimed)
|
|
proposal = SavingsProposal(**ir)
|
|
if status == "validated":
|
|
outcome: Any = ValidatedProposal(
|
|
proposal=proposal, p10=1.0, p50=2.0, p90=3.0, nominal_feasible=2.0
|
|
)
|
|
elif status == "unsupported":
|
|
outcome = Unsupported(
|
|
proposal=proposal,
|
|
reason=detail,
|
|
validated=ValidatedProposal(
|
|
proposal=proposal, p10=1.0, p50=2.0, p90=3.0, nominal_feasible=2.0
|
|
),
|
|
)
|
|
else:
|
|
outcome = Rejection(proposal=proposal, reason=detail)
|
|
write_outbox(
|
|
str(outbox),
|
|
run_id,
|
|
outcome=outcome,
|
|
provenance=_stamp("rejected" if status == "rejected" else "validated"),
|
|
checker_verdict="approve",
|
|
verdict_id=verdict_key(features_from_ir(ir)),
|
|
approach_id=aid,
|
|
)
|
|
assert nok is None or nok == claimed # the table's own consistency, not the builder's
|
|
write_coverage(str(outbox), run_id, rows=_coverage_rows(spec), stop_reason=_STOP_REASON)
|
|
return outbox
|
|
|
|
|
|
def _build(tmp_path: Path, **kwargs: Any) -> rb.Built:
|
|
outbox = kwargs.pop("outbox", None) or _outbox(tmp_path)
|
|
rounds = kwargs.pop("rounds_dir", None) or tmp_path / "runder"
|
|
return rb.build_round(
|
|
outbox, rounds, kwargs.pop("n", 0), ran_at=kwargs.pop("ran_at", _RAN_AT), **kwargs
|
|
)
|
|
|
|
|
|
def _feedback_file(path: Path, *ids: str) -> Path:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(
|
|
json.dumps(
|
|
{
|
|
"author": "fagpersonen",
|
|
"given_at": "2026-09-17T09:00:00+02:00",
|
|
"items": [
|
|
{"id": i, "type": 1, "text": f"Tallet i rad {i} ser for hoyt ut."} for i in ids
|
|
],
|
|
},
|
|
ensure_ascii=False,
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
return path
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------
|
|
# What the builder produces
|
|
# ---------------------------------------------------------------------------------------------
|
|
|
|
|
|
def test_the_round_has_exactly_the_four_parts_the_gate_names(tmp_path: Path) -> None:
|
|
"""The contract in ``v1_gate --help`` for round 0 is four names, and the builder owns three of
|
|
them. The fourth is the operator's and must be ABSENT from a freshly built round."""
|
|
built = _build(tmp_path)
|
|
round_dir = tmp_path / "runder" / "0"
|
|
assert built.round_dir == round_dir
|
|
assert (round_dir / "outcome.json").is_file()
|
|
assert (round_dir / "report.md").is_file()
|
|
assert (round_dir / "outbox").is_dir()
|
|
assert not (round_dir / _ATTEST_FILE).exists()
|
|
assert built.run_id == _RUN
|
|
|
|
|
|
def test_the_outbox_is_a_copy_that_no_longer_depends_on_the_source(tmp_path: Path) -> None:
|
|
"""MUTANT: link the outbox instead of copying it.
|
|
|
|
A link is refuted twice over — nothing under the round is a symlink (which is what
|
|
``gate.outbox_escape`` refuses), and DELETING the source afterwards leaves the round whole.
|
|
The second half is the one a ``copy`` that secretly links cannot survive."""
|
|
source = _outbox(tmp_path)
|
|
before = {p.name: p.read_bytes() for p in sorted(source.iterdir())}
|
|
built = _build(tmp_path, outbox=source)
|
|
outbox = built.round_dir / "outbox"
|
|
assert not outbox.is_symlink()
|
|
assert [p for p in sorted(outbox.rglob("*")) if p.is_symlink()] == []
|
|
shutil.rmtree(source)
|
|
after = {p.name: p.read_bytes() for p in sorted(outbox.iterdir())}
|
|
assert after == before
|
|
assert gate.outbox_escape(built.round_dir) == ""
|
|
|
|
|
|
def test_only_this_runs_artefacts_are_copied_and_the_rest_is_reported(tmp_path: Path) -> None:
|
|
"""A stress script leaves shell captures beside the run. They are not the run's artefacts, so
|
|
they do not enter the round — and they are NAMED in the result rather than dropped silently.
|
|
|
|
The expected copy list is counted from ``_SPEC``: two files per evaluated approach plus the
|
|
coverage."""
|
|
source = _outbox(tmp_path)
|
|
(source / "08.dry.out").write_text("stdout fra skriptet\n", encoding="utf-8")
|
|
(source / "verdict.err").write_text("", encoding="utf-8")
|
|
built = _build(tmp_path, outbox=source)
|
|
expected = {f"{_RUN}-coverage.json"} | {
|
|
f"{_RUN}-{aid}-{kind}.json" for aid in _EVALUATED for kind in ("proposal", "outcome")
|
|
}
|
|
assert len(expected) == 2 * len(_EVALUATED) + 1
|
|
assert set(built.copied) == expected
|
|
assert {p.name for p in (built.round_dir / "outbox").iterdir()} == expected
|
|
assert set(built.ignored) == {"08.dry.out", "verdict.err"}
|
|
|
|
|
|
def test_the_outcome_is_derived_row_for_row_from_the_runs_own_coverage(tmp_path: Path) -> None:
|
|
"""(a)-(d) per approach, counted from the table. ``feedback_ids`` is empty on every row: no
|
|
run records which feedback item produced which row, and the builder invents no tracking."""
|
|
built = _build(tmp_path)
|
|
outcome = json.loads((built.round_dir / "outcome.json").read_text(encoding="utf-8"))
|
|
assert outcome["run_id"] == _RUN
|
|
assert outcome["ran_at"] == _RAN_AT
|
|
assert outcome["removed"] == []
|
|
assert "outbox" not in outcome, "the gate refuses a round that names its own outbox"
|
|
rows = {row["id"]: row for row in outcome["approaches"]}
|
|
assert list(rows) == [aid for aid, *_ in _SPEC], "coverage's order, not a set's"
|
|
for aid, _label, status, nok, _claimed, detail in _SPEC:
|
|
row = rows[aid]
|
|
assert row["validated"] is (status == "validated"), aid
|
|
assert row["validated_nok"] == nok, aid
|
|
assert row["feedback_ids"] == [], aid
|
|
expected_stage = {"validated": "", "not_evaluated": "not_evaluated"}.get(
|
|
status, "unsupported" if detail == UNSUPPORTED_REASON else "stage4-p90"
|
|
)
|
|
assert row["stage"] == expected_stage, aid
|
|
|
|
|
|
def test_a_validated_row_the_runs_own_outcome_artefact_denies_is_refused(tmp_path: Path) -> None:
|
|
"""MUTANT: derive ``validated`` from coverage alone.
|
|
|
|
Coverage is the mandate's report; the per-approach outcome artefact is the validator's. When
|
|
the two disagree the run does not stand for itself, and the builder must say so by name rather
|
|
than pick the more flattering half. The rc-0 control is every other test in this file: the
|
|
same two sources AGREEING build a round."""
|
|
source = _outbox(tmp_path)
|
|
artefact = source / f"{_RUN}-{_FELL[0]}-outcome.json"
|
|
payload = json.loads(artefact.read_text(encoding="utf-8"))
|
|
payload["outcome_type"] = "validated"
|
|
artefact.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
_build(tmp_path, outbox=source)
|
|
assert _FELL[0] in str(caught.value)
|
|
assert "validated" in str(caught.value)
|
|
assert not (tmp_path / "runder" / "0").exists(), "a refusal leaves no half-built round"
|
|
|
|
|
|
def test_the_stray_artefact_is_refused_by_name_and_never_copied(tmp_path: Path) -> None:
|
|
"""MUTANT: copy every ``<run_id>-*.json`` and let the gate find the stray later.
|
|
|
|
An artefact of this run naming an approach the coverage does not list belongs to some other
|
|
run (``gate.verify_run``). Copying it in would hand the gate a round that is red for a reason
|
|
the builder already knew."""
|
|
source = _outbox(tmp_path)
|
|
shutil.copy(
|
|
source / f"{_RUN}-{_VALIDATED[0]}-proposal.json",
|
|
source / f"{_RUN}-a9-fra-en-annen-kjoring-proposal.json",
|
|
)
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
_build(tmp_path, outbox=source)
|
|
assert "a9-fra-en-annen-kjoring" in str(caught.value)
|
|
assert not (tmp_path / "runder" / "0").exists()
|
|
|
|
|
|
def test_an_outbox_without_coverage_says_which_flag_was_missing(tmp_path: Path) -> None:
|
|
"""``{run_id}-coverage.json`` is written IFF the run had both ``--mandate`` and
|
|
``--outbox-dir`` (``run.py``). Without it there is no list of commissioned approaches at all,
|
|
so the refusal has to name the cause and not merely the missing file."""
|
|
source = _outbox(tmp_path)
|
|
(source / f"{_RUN}-coverage.json").unlink()
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
_build(tmp_path, outbox=source)
|
|
assert "--mandate" in str(caught.value)
|
|
|
|
|
|
def test_two_coverage_files_are_two_runs_and_neither_is_the_round(tmp_path: Path) -> None:
|
|
"""A directory holding two runs cannot say which one the round presents, and guessing would
|
|
make the round's identity depend on sort order."""
|
|
source = _outbox(tmp_path)
|
|
shutil.copy(source / f"{_RUN}-coverage.json", source / "annen-kjoring-coverage.json")
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
_build(tmp_path, outbox=source)
|
|
assert "annen-kjoring" in str(caught.value)
|
|
|
|
|
|
def test_an_existing_round_is_never_overwritten(tmp_path: Path) -> None:
|
|
"""Round n holds the expert's own feedback and, later, the report they corrected. A second
|
|
build must refuse rather than quietly replace a round that has been read."""
|
|
built = _build(tmp_path)
|
|
(built.round_dir / "report.md").write_text("# min egen retting\n", encoding="utf-8")
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
_build(tmp_path)
|
|
assert str(built.round_dir) in str(caught.value)
|
|
assert (built.round_dir / "report.md").read_text(encoding="utf-8") == "# min egen retting\n"
|
|
|
|
|
|
def test_ran_at_must_carry_a_timezone(tmp_path: Path) -> None:
|
|
"""The gate reads ``ran_at`` as the run's time and refuses a stamp without a zone. Catching it
|
|
here costs one message instead of a round that builds and then reads unparseable."""
|
|
with pytest.raises(rb.RoundBuildError):
|
|
_build(tmp_path, ran_at="2026-09-18T11:20:00")
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------
|
|
# The report — the one artefact a person reads
|
|
# ---------------------------------------------------------------------------------------------
|
|
|
|
|
|
def _report(tmp_path: Path, **kwargs: Any) -> str:
|
|
return (_build(tmp_path, **kwargs).round_dir / "report.md").read_text(encoding="utf-8")
|
|
|
|
|
|
def test_the_report_names_every_approach_that_was_considered(tmp_path: Path) -> None:
|
|
"""(a) of ``PLAN.md § Målbar endring``. Counted from the table: every label appears, and the
|
|
count of labels present equals the number commissioned."""
|
|
text = _report(tmp_path)
|
|
present = [label for _aid, label, *_ in _SPEC if label in text]
|
|
assert len(present) == _COMMISSIONED, sorted({label for _a, label, *_ in _SPEC} - set(present))
|
|
|
|
|
|
def test_the_report_states_the_validated_saving_per_approach_and_in_total(tmp_path: Path) -> None:
|
|
"""(d). MUTANT: sum the wrong field.
|
|
|
|
The total is the VALIDATED rows' own amounts, quantized per amount (kø-(p)). Summing the
|
|
proposals' claims instead would reach 2 670 000,02, and summing the floats before converting
|
|
would reach 1 370 000,01 — both are spelled out here so the assert cannot pass by accident."""
|
|
built = _build(tmp_path)
|
|
assert built.validated_ore == _VALIDATED_ORE == 137_000_002
|
|
text = (built.round_dir / "report.md").read_text(encoding="utf-8")
|
|
assert "1 370 000,02" in text
|
|
assert "1 370 000,01" not in text, "the floats were summed before they were quantized"
|
|
assert "2 670 000,02" not in text, "the claims were summed instead of the validated amounts"
|
|
for _aid, label, status, nok, _claimed, _detail in _SPEC:
|
|
if status != "validated":
|
|
continue
|
|
assert label in text, label
|
|
whole, rest = divmod(to_ore(nok), 100)
|
|
assert f"{whole:,}".replace(",", " ") + f",{rest:02d}" in text, label
|
|
|
|
|
|
def test_the_report_says_where_and_why_each_refused_approach_fell(tmp_path: Path) -> None:
|
|
"""(c). MUTANT: drop a refused approach from the report.
|
|
|
|
Every approach that fell is named, with the stage in the expert's words AND the validator's
|
|
own sentence — the number of refusal sections is counted from the table, not from the text."""
|
|
text = _report(tmp_path)
|
|
for aid, label, status, _nok, _claimed, detail in _SPEC:
|
|
if status not in ("rejected", "unsupported"):
|
|
continue
|
|
assert label in text, aid
|
|
assert detail in text, aid
|
|
stage = "unsupported" if detail == UNSUPPORTED_REASON else "stage4-p90"
|
|
assert rb.STAGE_PROSE[stage] in text, aid
|
|
section = text.split("## Hva falt, og hvorfor", 1)[1].split("\n## ", 1)[0]
|
|
assert section.count("### ") == len(_FELL) == 2
|
|
|
|
|
|
def test_the_stage_vocabulary_covers_every_stage_the_validator_can_name(tmp_path: Path) -> None:
|
|
"""The prose table is pinned to the validator's OWN stage list. A stage added there without a
|
|
sentence here would otherwise reach a domain expert as a bare identifier."""
|
|
from portfolio_optimiser import validator
|
|
|
|
stages = {stage for stage, _markers in validator._REJECTION_STAGES}
|
|
assert stages | {"other", "not_evaluated"} == set(rb.STAGE_PROSE)
|
|
assert len(stages) == 6
|
|
assert all(len(text) > 20 for text in rb.STAGE_PROSE.values())
|
|
# Both halves are load-bearing: the short name before the colon is what a one-line diff of
|
|
# what changed can carry, and without it the diff falls back on the raw stage id.
|
|
assert all(": " in text for text in rb.STAGE_PROSE.values())
|
|
|
|
|
|
def test_the_report_carries_no_json_dump_and_no_unexplained_identifier(tmp_path: Path) -> None:
|
|
""" "Ingen JSON-dump, ingen interne id-er uten forklaring". The verdict key and the raw IR are
|
|
the two that would leak if the report were built by serialising the artefacts."""
|
|
text = _report(tmp_path)
|
|
assert "verdict_id" not in text
|
|
assert "claimed_saving_nok" not in text
|
|
assert not any(stage in text for stage in rb.STAGE_PROSE if stage.startswith("stage"))
|
|
assert '{"' not in text and "```json" not in text
|
|
|
|
|
|
def test_two_builds_of_the_same_outbox_are_byte_identical(tmp_path: Path) -> None:
|
|
"""MUTANT: make the section or row order depend on a set.
|
|
|
|
Row 4 counts content lines IN ORDER, so a report that reshuffles between builds would read as
|
|
an expert's edit. Both the whole-file bytes and the heading sequence are checked, because two
|
|
files can differ in bytes for a reason that is not order."""
|
|
source = _outbox(tmp_path)
|
|
first = _build(tmp_path, outbox=source, rounds_dir=tmp_path / "a").round_dir
|
|
second = _build(tmp_path, outbox=source, rounds_dir=tmp_path / "b").round_dir
|
|
for name in ("report.md", "outcome.json"):
|
|
assert (first / name).read_bytes() == (second / name).read_bytes(), name
|
|
headings = [
|
|
x
|
|
for x in (first / "report.md").read_text(encoding="utf-8").splitlines()
|
|
if x.startswith("## ")
|
|
]
|
|
assert headings == [
|
|
"## Hva ble vurdert",
|
|
"## Hva holdt, og hva det er verdt",
|
|
"## Hva falt, og hvorfor",
|
|
"## Hva kjøringen aldri rakk",
|
|
"## Slik leser du tallene",
|
|
]
|
|
|
|
|
|
def test_the_report_follows_the_runs_own_order_not_a_sorted_one(tmp_path: Path) -> None:
|
|
"""Determinism is not the same as sortedness: the run's coverage has an order, and that is the
|
|
order the expert reads. A reversed coverage must give a reversed report."""
|
|
reversed_spec = tuple(reversed(_SPEC))
|
|
source = _outbox(tmp_path / "rev", spec=reversed_spec)
|
|
considered = _report(tmp_path / "rev", outbox=source).split("## Hva ble vurdert", 1)[1]
|
|
considered = considered.split("\n## ", 1)[0]
|
|
at = [considered.index(label) for _a, label, *_ in reversed_spec]
|
|
assert at == sorted(at), "the report does not follow the coverage's own order"
|
|
forward = [considered.index(label) for _a, label, *_ in _SPEC]
|
|
assert forward == sorted(forward, reverse=True), "the report sorted instead of following it"
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------
|
|
# Round n >= 1 — measured against the round before it
|
|
# ---------------------------------------------------------------------------------------------
|
|
|
|
|
|
def _round_zero(tmp_path: Path) -> Path:
|
|
rounds = tmp_path / "runder"
|
|
_build(tmp_path, rounds_dir=rounds)
|
|
return rounds
|
|
|
|
|
|
def test_round_one_needs_the_experts_feedback_and_the_round_before_it(tmp_path: Path) -> None:
|
|
"""Round n is "feedback on report n-1, then run n". Without either half there is nothing to
|
|
measure the round against, and a round that measures against nothing is not a round."""
|
|
rounds = _round_zero(tmp_path)
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
rb.build_round(_outbox(tmp_path / "k1"), rounds, 1, ran_at=_RAN_AT)
|
|
assert "feedback.json" in str(caught.value)
|
|
feedback = _feedback_file(tmp_path / "f1.json", "f1")
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
rb.build_round(
|
|
_outbox(tmp_path / "k2"), tmp_path / "tomt", 1, ran_at=_RAN_AT, feedback=feedback
|
|
)
|
|
assert "runde 0" in str(caught.value)
|
|
|
|
|
|
def test_round_zero_refuses_a_feedback_file(tmp_path: Path) -> None:
|
|
"""Round 0 is the baseline: there is no report for anyone to have commented on yet."""
|
|
with pytest.raises(rb.RoundBuildError):
|
|
_build(tmp_path, feedback=_feedback_file(tmp_path / "f0.json", "f0"))
|
|
|
|
|
|
def test_round_one_copies_the_feedback_and_derives_what_was_removed(tmp_path: Path) -> None:
|
|
"""An approach round 0 had and round 1 does not is ``removed`` — DERIVED from the round
|
|
before, never declared. Its ``feedback_ids`` stays empty for the same reason every other row's
|
|
does: the run recorded no tracking, and the builder does not invent one."""
|
|
rounds = _round_zero(tmp_path)
|
|
shorter = tuple(row for row in _SPEC if row[0] != _VALIDATED[-1])
|
|
feedback = _feedback_file(tmp_path / "f1.json", "f1", "f2")
|
|
built = rb.build_round(
|
|
_outbox(tmp_path / "k1", spec=shorter), rounds, 1, ran_at=_RAN_AT, feedback=feedback
|
|
)
|
|
assert (built.round_dir / "feedback.json").read_bytes() == feedback.read_bytes()
|
|
outcome = json.loads((built.round_dir / "outcome.json").read_text(encoding="utf-8"))
|
|
assert outcome["removed"] == [{"id": _VALIDATED[-1], "feedback_ids": []}]
|
|
assert len(outcome["approaches"]) == _COMMISSIONED - 1
|
|
|
|
|
|
def test_round_one_reports_what_changed_and_admits_when_nothing_explains_it(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
""" "Endret siden forrige runde": every changed row is named, and a row no feedback id explains
|
|
SAYS so. Model noise has to be visible, not hidden — and today no run tracks feedback at all,
|
|
so every row lands on that branch and the report must not pretend otherwise."""
|
|
rounds = _round_zero(tmp_path)
|
|
flipped = _SPEC[0]
|
|
assert flipped[0] == _VALIDATED[0] and flipped[2] == "validated"
|
|
changed = ((flipped[0], flipped[1], "rejected", None, flipped[4], _SPEC[1][5]),) + _SPEC[1:]
|
|
built = rb.build_round(
|
|
_outbox(tmp_path / "k1", spec=changed),
|
|
rounds,
|
|
1,
|
|
ran_at=_RAN_AT,
|
|
feedback=_feedback_file(tmp_path / "f1.json", "f1"),
|
|
)
|
|
text = (built.round_dir / "report.md").read_text(encoding="utf-8")
|
|
section = text.split("## Endret siden forrige runde", 1)[1].split("\n## ", 1)[0]
|
|
assert flipped[1] in section
|
|
assert "ingen tilbakemelding forklarer dette" in section.lower()
|
|
assert "stage4-p90" not in section, "a raw stage id reached the expert"
|
|
assert rb.STAGE_PROSE["stage4-p90"].split(":")[0] in section
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------
|
|
# The one file the builder may never write
|
|
# ---------------------------------------------------------------------------------------------
|
|
|
|
|
|
def test_no_entry_point_leaves_an_attestation_behind(tmp_path: Path) -> None:
|
|
"""MUTANT: write the attestation so the round goes green.
|
|
|
|
Rows 1-2 are FORM OK until a PERSON says the round was held. A builder that could write that
|
|
file would put the gate's one un-computable step back inside the machine. Measured as
|
|
BEHAVIOUR over every entry point, because a grep for the name reads a name."""
|
|
rounds = tmp_path / "runder"
|
|
source = _outbox(tmp_path)
|
|
entries = {
|
|
"build_round": lambda: rb.build_round(source, rounds, 0, ran_at=_RAN_AT),
|
|
"read_run": lambda: rb.read_run(source),
|
|
"main": lambda: rb.main(
|
|
[
|
|
"--outbox",
|
|
str(source),
|
|
"--round",
|
|
"1",
|
|
"--rounds-dir",
|
|
str(rounds),
|
|
"--ran-at",
|
|
_RAN_AT,
|
|
"--feedback",
|
|
str(_feedback_file(tmp_path / "f1.json", "f1")),
|
|
]
|
|
),
|
|
}
|
|
for name, entry in entries.items():
|
|
entry()
|
|
assert list(rounds.rglob(_ATTEST_FILE)) == [], name
|
|
assert list(source.rglob(_ATTEST_FILE)) == [], name
|
|
|
|
|
|
def test_the_builder_writes_exactly_the_names_it_is_allowed_to(tmp_path: Path) -> None:
|
|
"""The set of paths a build creates, counted from ``_SPEC`` rather than listed by the builder.
|
|
|
|
Stronger than grepping for one forbidden name: ANY extra file — an attestation, a stray copy,
|
|
a scratch file left behind — makes this red, and the expected set is derivable without
|
|
reading the implementation."""
|
|
rounds = _round_zero(tmp_path)
|
|
feedback = _feedback_file(tmp_path / "f1.json", "f1")
|
|
built = rb.build_round(_outbox(tmp_path / "k1"), rounds, 1, ran_at=_RAN_AT, feedback=feedback)
|
|
made = {str(p.relative_to(built.round_dir)) for p in built.round_dir.rglob("*")}
|
|
expected = (
|
|
{"outcome.json", "report.md", "feedback.json", "outbox"}
|
|
| {f"outbox/{_RUN}-coverage.json"}
|
|
| {
|
|
f"outbox/{_RUN}-{aid}-{kind}.json"
|
|
for aid in _EVALUATED
|
|
for kind in ("proposal", "outcome")
|
|
}
|
|
)
|
|
assert made == expected
|
|
assert len(expected) == 4 + 1 + 2 * len(_EVALUATED)
|
|
|
|
|
|
# ---------------------------------------------------------------------------------------------
|
|
# End of the chain — the gate reads what the builder wrote
|
|
# ---------------------------------------------------------------------------------------------
|
|
|
|
|
|
def test_the_gate_reads_the_built_round_as_form_ok_and_not_as_unreadable(tmp_path: Path) -> None:
|
|
"""The whole point: round 0 can now be MADE. The gate must parse it, find nothing wrong with
|
|
its form, and still count it 0 — the step to green is the operator's attestation, which the
|
|
builder never writes. Both halves are asserted, because "unreadable" also counts 0."""
|
|
built = _build(tmp_path)
|
|
outcome, why = gate.read_outcome(built.round_dir)
|
|
assert outcome is not None, why
|
|
assert outcome.run_id == _RUN
|
|
assert gate.verify_run(built.round_dir / "outbox", _RUN, _coverage_rows()) == ""
|
|
attested = gate.read_attestation(built.round_dir)
|
|
assert not attested.ok and not attested.present, "the builder must leave row 2 un-green"
|
|
|
|
|
|
def test_the_command_line_builds_a_round_and_refuses_one_it_cannot(tmp_path: Path) -> None:
|
|
"""One command, as a SUBPROCESS: the exit codes are the operator's only signal, and 0 for a
|
|
refusal would be the worst of them. The rc-0 control comes first."""
|
|
source = _outbox(tmp_path)
|
|
rounds = tmp_path / "runder"
|
|
base = [sys.executable, "-m", "portfolio_optimiser.evals.round_builder"]
|
|
ok = subprocess.run(
|
|
[
|
|
*base,
|
|
"--outbox",
|
|
str(source),
|
|
"--round",
|
|
"0",
|
|
"--rounds-dir",
|
|
str(rounds),
|
|
"--ran-at",
|
|
_RAN_AT,
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
cwd=Path(rb.__file__).resolve().parents[3],
|
|
)
|
|
assert ok.returncode == 0, ok.stderr
|
|
assert (rounds / "0" / "report.md").is_file()
|
|
assert _RUN in ok.stdout
|
|
again = subprocess.run(
|
|
[
|
|
*base,
|
|
"--outbox",
|
|
str(source),
|
|
"--round",
|
|
"0",
|
|
"--rounds-dir",
|
|
str(rounds),
|
|
"--ran-at",
|
|
_RAN_AT,
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
cwd=Path(rb.__file__).resolve().parents[3],
|
|
)
|
|
assert again.returncode == 1, again.stdout
|
|
assert list(rounds.rglob(_ATTEST_FILE)) == []
|
|
|
|
|
|
def test_a_rounds_dir_inside_the_repo_must_be_gitignored(tmp_path: Path) -> None:
|
|
"""The remote is public and a domain expert's feedback must never reach it. The builder
|
|
CREATES the directory the gate only reads, so it is the place the check has to hold."""
|
|
repo = Path(rb.__file__).resolve().parents[3]
|
|
assert gate.safe_rounds_dir(repo / gate.DEFAULT_ROUNDS_DIR)
|
|
assert not gate.safe_rounds_dir(repo / "docs" / "ikke-ignorert-runder")
|
|
with pytest.raises(rb.RoundBuildError) as caught:
|
|
_build(tmp_path, rounds_dir=repo / "docs" / "ikke-ignorert-runder")
|
|
assert "gitignore" in str(caught.value).lower()
|
|
assert not (repo / "docs" / "ikke-ignorert-runder").exists()
|
|
|
|
|
|
def test_the_builder_makes_no_network_call_and_reads_no_clock(tmp_path: Path) -> None:
|
|
"""Offline and wall-clock-free by construction: the module imports no socket and no
|
|
``datetime.now``/``time.time``, so "no network, no clock" is a property of the text rather
|
|
than of the run that happened to be measured."""
|
|
source = Path(rb.__file__).read_text(encoding="utf-8")
|
|
for banned in ("import socket", "urllib", "requests", "datetime.now", "time.time", "utcnow"):
|
|
assert banned not in source, banned
|
|
assert "os.environ" not in source
|
|
|
|
|
|
def test_the_help_text_states_the_shape_it_writes(tmp_path: Path) -> None:
|
|
"""A command whose output another command must read has to say so where the operator looks."""
|
|
assert "outcome.json" in rb.ROUND_BUILD_CONTRACT
|
|
assert "report.md" in rb.ROUND_BUILD_CONTRACT
|
|
assert _ATTEST_FILE.split(".")[0] in rb.ROUND_BUILD_CONTRACT.lower()
|
|
out = subprocess.run(
|
|
[sys.executable, "-m", "portfolio_optimiser.evals.round_builder", "--help"],
|
|
capture_output=True,
|
|
text=True,
|
|
cwd=Path(rb.__file__).resolve().parents[3],
|
|
)
|
|
assert out.returncode == 0
|
|
assert "outcome.json" in out.stdout
|