"""S2.0 load-bearing seam (målbilde §1/§5/§7): the portfolio must learn ACROSS projects — a verdict captured on project *k* MUST reach project *k+1*'s hypothesis prompt. Before Fase 2a, ``run_portfolio`` called ``run_project`` WITHOUT ``bundle_dir``/``verdict_dir``, so the ``bundle_dir``-gated Step-1 ExpeL fold never fired on the portfolio path and the store-threaded verdict never reached the next hypothesis — an honesty breach, since ``run_portfolio``'s docstring already claimed a "cross-project learning loop". Topology (road-*k* + bundle-*(k+1)*, ONE bundle): project *k* is road-backed (``docs_dir``, ``bundle_dir=None``); project *k+1* is bundle-backed (``bundle_dir`` = the repo-local mini-bundle, ``id`` = its IR ``project_id``). *k*'s scripted proposal shares *k+1*'s candidate features (code ``ENERGI-TOTAL-EL``, magnitude ~18000), so *k*'s captured verdict is what *k+1*'s ExpeL query retrieves. The marker is a SENTINEL absent from the bundle — its presence in *k+1*'s prompt can only come from *k*'s verdict via the threaded store (detach the threading in run.py:496 → RED). """ from __future__ import annotations from importlib.resources import files import pytest from portfolio_optimiser.contracts import GoalContract from portfolio_optimiser.reference_domain import Project from portfolio_optimiser.run import _goal_limit_if_reached, run_portfolio from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict _MINI_BUNDLE = str(files("portfolio_optimiser").joinpath("data/bundles/bygg-energi-mikro-a")) # A VALID proposal whose features align with the mini-bundle candidate (code ENERGI-TOTAL-EL, # magnitude ~18000): affected total 180000 x 1.0, no assumptions -> degenerate Monte Carlo # P90 = 0.30 x 180000 = 54000 >= claimed 18000 -> validates on the first attempt. _ALIGNED_REPLY = ( '{"measure":"LED-retrofit av kontorbelysning","affected_items":' '[{"code":"ENERGI-TOTAL-EL","quantity":180000,"unit_cost":1.0}],"claimed_saving_nok":18000}' ) # Markers absent from the bundle: presence in a prompt/store can only come via the threaded seams. _SENTINEL = "SENTINEL-XPROJ-4a9f2b realiseringskorreksjon fra prosjekt k" _INBOX_SENTINEL = "SENTINEL-INBOX-7c3e1d dropped-after-run realiseringskorreksjon" def _generation_prompts(sink: list[str]) -> list[str]: """The generation-call prompts (``generate._build_messages`` embeds 'SavingsProposal'), isolated from the debate-round prompts also captured in the shared sink.""" return [p for p in sink if "SavingsProposal" in p] def _make_docs(tmp_path, name: str) -> str: """A tmp docs folder with citable content, so the road path retrieval is non-empty.""" d = tmp_path / name d.mkdir() (d / "cost.txt").write_text( "Asphalt Ab11 unit rate renegotiation reduced the paving cost on the school stretch.", encoding="utf-8", ) return str(d) def _road_k(tmp_path, *, rationale: str) -> Project: return Project( id="ROAD-K", name="Road k", description="road-backed project k", currency="NOK", cost_items=(), docs_dir=_make_docs(tmp_path, "k-docs"), verdict_input={"decision": "approved", "rationale": rationale}, bundle_dir=None, verdict_dir=None, ) def _bundle_kplus1(tmp_path, *, verdict_dir: str | None = None) -> Project: return Project( id="BYGG-ENERGI-MIKRO-A", name="Bundle k+1", description="bundle-backed project k+1", currency="NOK", cost_items=(), docs_dir=_make_docs(tmp_path, "kplus1-docs"), verdict_input={"decision": "approved", "rationale": "k+1 reviewed (sim)"}, bundle_dir=_MINI_BUNDLE, verdict_dir=verdict_dir, ) async def test_verdict_on_k_reaches_kplus1_hypothesis_prompt( tmp_path, monkeypatch, make_recording_client_factory ) -> None: """T-2.0a LOAD-BEARING: a verdict captured on road-backed project *k* reaches bundle-backed *k+1*'s hypothesis-generation prompt (both the sentinel rationale and *k*'s verdict id, via the ExpeL few-shot). Detach ``bundle_dir=project.bundle_dir`` at run.py:496 → *k+1* runs the road path → the Step-1 fold is skipped → the sentinel never arrives → RED.""" k = _road_k(tmp_path, rationale=_SENTINEL) kplus1 = _bundle_kplus1(tmp_path) monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (k, kplus1)) factory, recorded = make_recording_client_factory(_ALIGNED_REPLY) result = await run_portfolio(profile="local", client_factory=factory) # k ran first and its captured verdict carries the sentinel (its rationale is verbatim). assert result.runs[0].verdict.rationale == _SENTINEL k_verdict_id = result.runs[0].verdict.id gen_prompts = _generation_prompts(recorded) assert gen_prompts, "the generation calls must have happened" assert any(_SENTINEL in p for p in gen_prompts), ( "k's verdict rationale did not reach k+1's hypothesis prompt — the bundle_dir threading in " "run_portfolio is detached (no cross-project ExpeL fold)" ) assert any(k_verdict_id in p for p in gen_prompts), ( "k's verdict id did not reach the hypothesis prompt via the ExpeL few-shot" ) async def test_kplus1_alone_against_fresh_store_carries_no_signal( tmp_path, monkeypatch, make_recording_client_factory ) -> None: """T-2.0b CAUSALITY CONTROL: running ONLY *k+1* against a fresh (empty) store carries the sentinel into NO prompt — proving the signal in T-2.0a is caused by the threaded store, not incidental to the fixture (makes the positive test genuinely load-bearing).""" k = _road_k(tmp_path, rationale=_SENTINEL) kplus1 = _bundle_kplus1(tmp_path) monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (k, kplus1)) factory, recorded = make_recording_client_factory(_ALIGNED_REPLY) # project_ids restricts the pass to k+1 alone; the default store is empty. await run_portfolio(["BYGG-ENERGI-MIKRO-A"], "local", client_factory=factory) assert all(_SENTINEL not in p for p in recorded), ( "the sentinel reached a prompt with no project-k verdict in the store — the positive test " "is not actually load-bearing" ) async def test_dropped_verdict_in_kplus1_inbox_reaches_store( tmp_path, monkeypatch, make_recording_client_factory ) -> None: """T-2.0d LOAD-BEARING (verdict_dir threading): a verdict dropped as a FILE in *k+1*'s ``verdict_dir`` inbox (an expert/persona authoring after an earlier run) is merged into the store on the portfolio path. Detach ``verdict_dir=project.verdict_dir`` at run.py:496 → the file is never read → the dropped verdict never reaches the store → RED. Reuses the Step-7-proven ``run_project`` merge (run.py:254-257) at the portfolio level.""" inbox = tmp_path / "kplus1-inbox" inbox.mkdir() # DISTINCT features from k's aligned proposal, so the dropped verdict's content-hash id does NOT # collide with k's captured verdict — otherwise the store's first-write-wins would drop it and # the test would fail for the wrong reason (id collision, not a detached seam). dropped = capture_verdict( ProposalFeatures( affected_codes=frozenset({"INBOX-ONLY-XY"}), measure_type="inbox-only measure (post-run drop)", claimed_saving_nok=99000.0, ), "approved", _INBOX_SENTINEL, ) write_verdict(str(inbox), dropped) k = _road_k(tmp_path, rationale=_SENTINEL) kplus1 = _bundle_kplus1(tmp_path, verdict_dir=str(inbox)) monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (k, kplus1)) factory, _recorded = make_recording_client_factory(_ALIGNED_REPLY) result = await run_portfolio(profile="local", client_factory=factory) assert any(_INBOX_SENTINEL in v.rationale for v in result.store.verdicts), ( "a verdict dropped as a file in k+1's verdict_dir inbox did not reach the store — the " "verdict_dir threading in run_portfolio is detached" ) def test_percent_goal_against_zero_baseline_raises_value_error() -> None: """T-2.0c (direct unit test of the chokepoint): a percent goal against a non-positive baseline is meaningless — ``int(percent/100 * 0) == 0`` would falsely read as 'goal reached'. The guard in ``_goal_limit_if_reached`` raises ``ValueError`` instead. Detach the guard → no raise (false 'reached') → RED. Tested DIRECTLY on the function: ``_goal_limit_if_reached`` is only called inside a non-empty loop with ``baseline=sum(total_cost)``, and no reference project has zero cost, so a baseline-0 path is not constructable via ``run_portfolio`` — the direct call is the unambiguous, constructable test.""" goal = GoalContract(percent=10.0) with pytest.raises(ValueError): _goal_limit_if_reached(goal, observed_ore=0, baseline_ore=0)