"""U4 + U13, part 2 — the CALL SITES. Load-bearing proofs for the seams econ 56 left open. Three things are proved here, and each of them is a seam a mutation can detach. **1. The trace is a CALLER-OWNED accumulator, for the reason the parse-failure sink is one (Fase 1b, funn 1).** ``explore()`` raises ``BudgetExceeded`` on its round cap, and a token cap fires from inside the middleware mid-run — on both paths ``ExplorationResult`` never returns, so a ``ledger_log`` that existed only as a return value would be destroyed by exactly the endings § C.2 requires the artefact to be readable after ("så en stoppet utforskning er lesbar uansett hvilken vakt som fyrte"). The accumulator the caller holds survives however the loop ended. ``ExplorationResult.ledger_log`` is BUILT FROM that accumulator rather than alongside it: two lists holding one fact is the kø-(p) drift class, one layer up. **2. The CLI door refuses everything it cannot honour, by name.** ``--explore`` and ``--mandate`` are two sources of ONE mandate and are REFUSED together rather than merged: ``explore()`` sets the objective from the prompt and hardcodes ``allow_own_proposals=True``, so composing them would silently overwrite three fields an operator wrote by hand. The refusal names the library API (``explore(seed_approaches=…)``) because § C.6 door 1 is a real need this surface does not serve. **3. ``{run_id}-exploration.json`` is written from a ``finally``**, so the run that most needs the evidence — the one a cap cut short — is the one that has it. The client is the repo's own ``ScriptedChatClient`` throughout (a bare ``BaseChatClient`` no-ops ``BudgetMiddleware``), and every tool assertion calls the tool's ``func`` DIRECTLY: measured in econ 56, a scripted run returns TEXT and never emits a tool call, so no scripted exploration reaches a tool body and a gate that only drove ``explore()`` would be vacuous. """ from __future__ import annotations import json from collections.abc import Callable from pathlib import Path from typing import Any import pytest from agent_framework import BaseChatClient import portfolio_optimiser from portfolio_optimiser import explore, okf, run from portfolio_optimiser.budget import BudgetExceeded from portfolio_optimiser.explore import ExplorationContract, ExplorationTrace from portfolio_optimiser.mandate import Approach, Mandate from portfolio_optimiser.simulation import ScriptedChatClient _BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro" _PID = "BYGG-KONTOR-NORD" #: The hypothesiser's marked line. The label is the marker the end-to-end arm looks for on stdout: #: it appears nowhere in the bundle, in the reference projects or in any other test, so its presence #: in the settlement can only have come through the mandate the exploration shaped. _LABEL = "SENTINEL-EXPLORE-7c1d33" _ENERGY_REPLY = ( '{"measure":"LED-retrofit av kontorbelysning","affected_items":' '[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}' ) _CONTRACT_JSON: dict[str, Any] = { "max_rounds": 4, "max_tokens": 100_000, "max_stall_count": 2, "max_reset_count": 1, "max_plan_revisions": 0, "enable_plan_review": False, } @pytest.fixture(autouse=True) def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None: """Hermetic env: no arm here may read the operator's Foundry configuration.""" monkeypatch.delenv("PORTFOLIO_MODEL_MAP", raising=False) monkeypatch.delenv("PORTFOLIO_FOUNDRY_PROJECT_ENDPOINT", raising=False) def _ledger_json(*, satisfied: bool, speaker: str = "hypothesiser") -> str: return json.dumps( { "is_request_satisfied": {"reason": "r", "answer": satisfied}, "is_in_loop": {"reason": "r", "answer": False}, "is_progress_being_made": {"reason": "r", "answer": True}, "next_speaker": {"reason": "r", "answer": speaker}, "instruction_or_question": {"reason": "r", "answer": "Shape one hypothesis."}, } ) def _manager_script(ledgers: list[str]) -> Callable[[str, str], str]: """Route a manager prompt blob to its scripted reply (the econ-56 helper, verbatim in shape). The stage ORDER is load-bearing (§ F, A6): the selector sees the CONCATENATION of the call's messages, so a later-stage prompt still carries the earlier stage's text. """ def _select(blob: str, _role: str) -> str: if "provide the final answer" in blob: return "FINAL: exploration done." if "pure JSON format" in blob: return ledgers.pop(0) if ledgers else _ledger_json(satisfied=True) if "went wrong on this last run" in blob: return "PLAN-UPDATE: revised plan." if "rewrite the following fact sheet" in blob: return "FACTS-UPDATE: revised facts." if "bullet-point plan" in blob: return "PLAN: - ask the hypothesiser" if "pre-survey" in blob: return "FACTS: the bundle is anchored." return "{}" return _select def _hypothesis_line(label: str, rationale: str) -> str: return f"{explore.HYPOTHESIS_MARKER} " + json.dumps({"label": label, "rationale": rationale}) def _factory( *, ledgers: list[str], hypothesiser: list[str], fallback: str = "ok" ) -> Callable[[str], BaseChatClient]: """One fresh ``ScriptedChatClient`` per role — the exploration's three plus everyone else. ``fallback`` serves the roles the PIPELINE builds (proposer/checker), so one factory can drive an exploration and the run it hands its mandate to. That is what makes the end-to-end arm an end-to-end arm rather than two half-proofs. """ def factory(role: str) -> BaseChatClient: if role == explore.MANAGER_ROLE: return ScriptedChatClient(reply_selector=_manager_script(ledgers), role=role) if role == explore.HYPOTHESISER_ROLE: replies = list(hypothesiser) def _hyp(_blob: str, _role: str) -> str: return replies.pop(0) if replies else "nothing further." return ScriptedChatClient(reply_selector=_hyp, role=role) if role == explore.NAVIGATOR_ROLE: return ScriptedChatClient("NAVIGATOR: index read.", role=role) return ScriptedChatClient(fallback, role=role) return factory def _contract(**overrides: Any) -> ExplorationContract: return ExplorationContract(**{**_CONTRACT_JSON, **overrides}) # --------------------------------------------------------------------------------------------- # 1. The caller-owned accumulator (the funn-1 sink shape, applied to the exploration) # --------------------------------------------------------------------------------------------- def _micro_bundle_dir() -> str: return str( Path(portfolio_optimiser.__file__).parent / "data" / "bundles" / "bygg-energi-baseline-mikro" ) def test_quick_validate_verdicts_reach_the_callers_trace() -> None: """T1: every advisory verdict the hypothesiser asked for is recorded where the caller can read it — the ONE thing ``ExplorationResult`` deliberately does not carry. Called DIRECTLY, because a scripted run never reaches a tool body (measured, econ 56): an arm that drove ``explore()`` and then asserted on an empty list would be green against every implementation, including one with no sink at all. Detach point: drop the ``sink`` append in ``quick_validate_tool`` → RED. """ base = _micro_bundle_dir() projection = dict(okf.load_ir_projection(base)) projection.pop("_note", None) trace = ExplorationTrace() validate = explore.quick_validate_tool((base,), sink=trace.quick_validations) honest = validate.func( bundle_id="bygg-energi-baseline-mikro", proposal_json=json.dumps(projection) ) invented = dict(projection) invented["affected_items"] = [ {**dict(projection["affected_items"][0]), "code": "CODE-THAT-DOES-NOT-EXIST"} ] validate.func(bundle_id="bygg-energi-baseline-mikro", proposal_json=json.dumps(invented)) assert len(trace.quick_validations) == 2, "both calls must be recorded, in call order" first, second = trace.quick_validations assert first.bundle_id == "bygg-energi-baseline-mikro" assert first.verdict == honest, "the recorded verdict must be the one the tool ANSWERED" assert second.verdict["decision"] == "rejected" assert "CODE-THAT-DOES-NOT-EXIST" in second.proposal_json @pytest.mark.asyncio async def test_the_returned_ledger_log_is_the_traces_own_entries() -> None: """T2: ``ExplorationResult.ledger_log`` is BUILT FROM the accumulator, never alongside it. Two lists holding one fact drift (kø-(p)), and a drifted pair would let the returned result and the written artefact describe different runs. Detach point: accumulate rounds in a second local list → RED. """ trace = ExplorationTrace() result = await explore.explore( "Find a saving.", contract=_contract(), bundle_dirs=(str(_BUNDLE_DIR),), client_factory=_factory( ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=True)], hypothesiser=[_hypothesis_line(_LABEL, "because the bundle says so")], ), trace=trace, ) assert result.stop is None assert len(trace.ledger) == 2 assert tuple(trace.ledger) == result.ledger_log @pytest.mark.asyncio async def test_the_trace_survives_the_budget_exception_that_destroys_the_result() -> None: """T3: the round cap raises, and the caller STILL holds every round the loop recorded. This is the whole reason the accumulator is caller-owned. The round cap leaves as a typed ``BudgetExceeded`` (econ 56), so nothing is returned — and § C.2 requires the artefact to be readable no matter which guard fired. Detach point: return the log only, keeping no caller-visible accumulator → RED. """ trace = ExplorationTrace() with pytest.raises(BudgetExceeded) as excinfo: await explore.explore( "Find a saving.", contract=_contract(max_rounds=2), bundle_dirs=(str(_BUNDLE_DIR),), client_factory=_factory( ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=False)], hypothesiser=["still thinking."], ), trace=trace, ) assert excinfo.value.kind == "exploration_rounds" assert len(trace.ledger) == 2, ( "the rounds the exploration DID record were destroyed with the result — the artefact a " "capped run needs most would be empty" ) # --------------------------------------------------------------------------------------------- # 2. The CLI door — every refusal by name, never a silent merge or a silent drop # --------------------------------------------------------------------------------------------- def _config_file(tmp_path: Path, **overrides: Any) -> str: path = tmp_path / "exploration.json" path.write_text(json.dumps({**_CONTRACT_JSON, **overrides}), encoding="utf-8") return str(path) def _mandate_file(tmp_path: Path) -> str: path = tmp_path / "mandate.json" path.write_text( json.dumps( { "objective": "cut energy cost", "approaches": [{"id": "a1", "label": "LED", "description": "swap the fittings"}], "allow_own_proposals": False, } ), encoding="utf-8", ) return str(path) def _base_argv(tmp_path: Path) -> list[str]: return [ _PID, "--docs-dir", str(_BUNDLE_DIR), "--bundle-dir", str(_BUNDLE_DIR), "--explore", "Find the cheapest saving.", "--explore-config", _config_file(tmp_path), ] def test_an_exploration_config_without_an_exploration_is_refused_by_name(tmp_path, capsys) -> None: """T4: ``--explore-config`` alone would be loaded and then dropped on the floor — the exact silent-ignore ``--embedder-config requires --semantic-retrieval`` exists to prevent. Detach point: drop the refusal → RED. """ rc = run.main( [_PID, "--docs-dir", str(_BUNDLE_DIR), "--explore-config", _config_file(tmp_path)] ) assert rc == 1 assert "--explore-config" in capsys.readouterr().err def test_an_exploration_without_its_bounds_is_refused_rather_than_defaulted( tmp_path, capsys ) -> None: """T5: ``--explore`` alone is refused — the CLI may not invent bounds. Every ``ExplorationContract`` field is required WITHOUT a default precisely because ``MagenticBuilder`` falls back to unbounded, and a CLI that supplied its own numbers would undo that decision one layer up. """ rc = run.main( [_PID, "--docs-dir", str(_BUNDLE_DIR), "--bundle-dir", str(_BUNDLE_DIR), "--explore", "go"] ) assert rc == 1 assert "--explore-config" in capsys.readouterr().err def test_explore_and_mandate_are_two_sources_of_one_mandate_and_are_refused_together( tmp_path, capsys ) -> None: """T6: the decision, made deliberately and stated: REFUSE, never merge. ``explore()`` takes the objective from the prompt and hardcodes ``allow_own_proposals=True``, so composing the two would silently overwrite fields the operator wrote by hand. The message names the library door (``seed_approaches``) so the refusal teaches instead of only forbidding. Detach point: let one source silently win → RED. """ rc = run.main(_base_argv(tmp_path) + ["--mandate", _mandate_file(tmp_path)]) assert rc == 1 err = capsys.readouterr().err assert "--explore" in err and "--mandate" in err assert "seed_approaches" in err, "the refusal must name the door that DOES serve door 1" def test_explore_and_live_dry_run_contradict_and_are_refused(tmp_path, capsys) -> None: """T7: ``--live-dry-run`` stops before the first model call; an exploration IS model calls.""" rc = run.main(_base_argv(tmp_path) + ["--live-dry-run"]) assert rc == 1 assert "--live-dry-run" in capsys.readouterr().err def test_an_exploration_with_no_knowledge_base_is_refused(tmp_path, capsys) -> None: """T8: without ``--bundle-dir`` the navigator has nothing to open — the loop would run, cost tokens and read nothing. Refused rather than run empty (the ``--semantic-retrieval`` shape). Detach point: drop the requirement → RED. """ rc = run.main( [ _PID, "--docs-dir", str(_BUNDLE_DIR), "--explore", "go", "--explore-config", _config_file(tmp_path), ] ) assert rc == 1 assert "--bundle-dir" in capsys.readouterr().err def test_a_plan_review_nobody_can_answer_is_refused_at_the_cli(tmp_path, capsys) -> None: """T9: ``enable_plan_review`` is the U13 SYNCHRONOUS door and this surface has no reviewer. Refused HERE rather than left to ``explore()``: ``ExplorationError`` is a ``RuntimeError``, so it is outside ``main()``'s ``(ValueError, FileNotFoundError, ValidationError)`` refusal tuple and would leave as a traceback instead of the rc-1 line every other misconfiguration produces. Detach point: let the flag through to ``explore()`` → RED (traceback, not rc 1). """ rc = run.main( _base_argv(tmp_path)[:-1] + [_config_file(tmp_path, enable_plan_review=True)], ) assert rc == 1 assert "enable_plan_review" in capsys.readouterr().err def test_explore_belongs_to_single_project_mode(tmp_path, capsys) -> None: """T10: portfolio mode is a documented partition, and ``--explore`` is on the single-project side of it — one exploration shapes ONE mandate against ONE knowledge base. The assertion names ``--portfolio``, and that was MEASURED rather than chosen: asserting only that the message mentions ``--explore`` passed against an implementation with no partition entry at all, because the run then fell through to ``--explore requires --bundle-dir``, which names ``--explore`` too. Two refusals sharing a substring is this repo's "assert never on wording two branches share" rule, caught by its own mutation. Detach point: drop ``--explore`` from the portfolio ``single_only`` partition → RED. """ rc = run.main(["--portfolio", "--explore", "go", "--explore-config", _config_file(tmp_path)]) assert rc == 1 err = capsys.readouterr().err assert "--explore" in err and "--portfolio" in err @pytest.fixture() def _explored_main(monkeypatch: pytest.MonkeyPatch) -> None: """Inject the role-dispatching scripted factory into the seam ``main()`` resolves through. ``main()`` passes no ``client_factory``, and ``explore()`` imports ``run._default_factory`` lazily at call time, so this ONE patch covers both the exploration and the pipeline it feeds — which is what makes the arm below end-to-end rather than a wiring spy. """ factory = _factory( ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=True)], hypothesiser=[_hypothesis_line(_LABEL, "the index says the fittings are old")], fallback=_ENERGY_REPLY, ) monkeypatch.setattr("portfolio_optimiser.run._default_factory", lambda profile: factory) def test_the_shaped_mandate_reaches_the_pipeline(tmp_path, capsys, _explored_main) -> None: """T11: the approach the hypothesiser shaped is SETTLED by the run — the whole point of (1). The settlement is printed only for a run that HAS a mandate, and the label appears nowhere in the bundle or the reference projects, so it can have reached stdout only by travelling prompt → ``explore()`` → ``Mandate`` → ``run_project(mandate=…)`` → ``settle``. Detach point: drop ``mandate=`` from the exploring branch's ``run_project`` call → RED. """ rc = run.main(_base_argv(tmp_path)) assert rc == 0 out = capsys.readouterr().out assert _LABEL in out, "the exploration's mandate never reached the pipeline's settlement" # --------------------------------------------------------------------------------------------- # 3. The artefact — written from a ``finally``, because a capped run is what it exists for # --------------------------------------------------------------------------------------------- def test_the_exploration_artefact_carries_the_rounds_and_the_advisory_verdicts( tmp_path, _explored_main ) -> None: """T12: ``{run_id}-exploration.json`` holds the per-round ledger AND the ``quick_validate`` verdicts — the level-1 evidence ``ExplorationResult`` deliberately does not carry (§ C.2). Detach point: drop the artefact write → RED. """ outbox = tmp_path / "outbox" rc = run.main(_base_argv(tmp_path) + ["--outbox-dir", str(outbox), "--run-id", "r1"]) assert rc == 0 payload = json.loads((outbox / "r1-exploration.json").read_text(encoding="utf-8")) assert payload["run_id"] == "r1" assert payload["completed"] is True assert payload["stop"] is None assert [row["round_index"] for row in payload["rounds"]] == [1, 2] assert payload["rounds"][-1]["is_request_satisfied"] is True assert payload["rounds"][0]["next_speaker"] == "hypothesiser" assert "quick_validations" in payload def test_the_artefact_is_written_even_when_the_exploration_was_cut_short( tmp_path, monkeypatch ) -> None: """T13: a capped exploration is the run whose evidence matters MOST, and it is the one that returns nothing — so the write lives in a ``finally`` (the ``write_parse_failures`` precedent). ``completed`` is a field rather than an inference: with no result there is no ``stop``, and a ``stop: null`` that meant BOTH "concluded normally" and "we never found out" would be the kind of silence this repo writes required fields to close. Detach point: move the write out of the ``finally`` → RED. """ factory = _factory( ledgers=[_ledger_json(satisfied=False), _ledger_json(satisfied=False)], hypothesiser=["still thinking."], fallback=_ENERGY_REPLY, ) monkeypatch.setattr("portfolio_optimiser.run._default_factory", lambda profile: factory) outbox = tmp_path / "outbox" with pytest.raises(BudgetExceeded): run.main( [ _PID, "--docs-dir", str(_BUNDLE_DIR), "--bundle-dir", str(_BUNDLE_DIR), "--explore", "go", "--explore-config", _config_file(tmp_path, max_rounds=2), "--outbox-dir", str(outbox), "--run-id", "r2", ] ) payload = json.loads((outbox / "r2-exploration.json").read_text(encoding="utf-8")) assert payload["completed"] is False assert payload["stop"] is None assert len(payload["rounds"]) == 2 def test_the_artefact_payload_is_byte_deterministic() -> None: """T14 (control): the same trace renders the same bytes, so the artefact is diff-stable like every other outbox file. Drives the renderer directly — the CLI arms above prove it is CALLED, this proves what it produces.""" trace = ExplorationTrace() trace.ledger.append( explore.LedgerEntry( round_index=1, is_request_satisfied=True, is_in_loop=False, is_progress_being_made=True, next_speaker="hypothesiser", instruction_or_question="Shape one hypothesis.", speaker_known=True, ) ) trace.quick_validations.append( explore.QuickValidation( bundle_id="b", proposal_json="{}", verdict={"decision": "unparseable"} ) ) first = explore.trace_payload(trace, stop=None, completed=True) second = explore.trace_payload(trace, stop=None, completed=True) assert json.dumps(first, sort_keys=True) == json.dumps(second, sort_keys=True) def test_a_seeded_mandate_still_leads_the_shaped_one() -> None: """T15 (control for T6's refusal): the library door the refusal names actually works. A refusal that pointed at a door which did not open would be worse than no message at all. """ seed = Approach(id="expert-1", label="expert's own", description="the domain expert asked") minted = explore._mint_approaches((seed,), [(_LABEL, "shaped in the loop")]) assert [a.id for a in minted] == ["expert-1", "hypothesis-1"] assert isinstance(Mandate(objective="o", approaches=minted, allow_own_proposals=True), Mandate)