"""S5.3 CLI-parity tests for ``run.main()`` single-project flags (Steps 2/4/5). In-process ``run.main([argv])`` + rc + ``capsys`` substring asserts (never subprocess), mirroring ``tests/test_live_dry_run.py``. Every arm is offline — it stops before the first model call (``debate.run``), so no socket/network is exercised (brief NFR). The bundle fixture ``shared/examples/bygg-energi-mikro`` (project ``BYGG-KONTOR-NORD``) supplies citable content so the dry-run reaches its offline return. """ from __future__ import annotations import json from pathlib import Path import pytest from portfolio_optimiser import run from portfolio_optimiser.dimension import Dimension from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger from portfolio_optimiser.run import run_project from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro" _PID = "BYGG-KONTOR-NORD" @pytest.fixture(autouse=True) def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None: """Hermetic env (verbatim from ``test_live_dry_run.py``): clear the S4.1 out-of-tree overrides so these CLI assertions read the BUNDLED map/config, not the operator's Foundry environment.""" monkeypatch.delenv("PORTFOLIO_MODEL_MAP", raising=False) monkeypatch.delenv("PORTFOLIO_FOUNDRY_PROJECT_ENDPOINT", raising=False) def _write_dimension(tmp_path: Path) -> Path: """A valid dimension scope config on disk (loaded fail-fast by ``load_dimension``).""" dim = Dimension( id="energi", label="Energi", allowed_measure_types=frozenset({"energy_efficiency"}), ) p = tmp_path / "dim.json" p.write_text(dim.model_dump_json(), encoding="utf-8") return p # --- Step 2: --dimension-config / --outbox-dir / --run-id wiring + structured refusal ------------- def test_dimension_config_flag_parses_offline(tmp_path, capsys) -> None: """(a) ``--dimension-config `` + ``--live-dry-run`` → rc 0 (flag parsed, loader invoked, offline — stops before any model call).""" rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--dimension-config", str(_write_dimension(tmp_path)), "--live-dry-run", ] ) assert rc == 0 assert "LIVE-DRY-RUN OK" in capsys.readouterr().out def test_outbox_dir_with_run_id_writes_runconfig_offline(tmp_path) -> None: """(b) ``--outbox-dir`` + ``--run-id`` + ``--live-dry-run`` → rc 0 AND ``/r1-runconfig.json`` written offline (via ``write_run_config``, before the dry-run return).""" outbox = tmp_path / "out" outbox.mkdir() rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--outbox-dir", str(outbox), "--run-id", "r1", "--live-dry-run", ] ) assert rc == 0 assert (outbox / "r1-runconfig.json").is_file() def test_outbox_dir_without_run_id_refuses(tmp_path, capsys) -> None: """(c) RED guard: ``--outbox-dir`` WITHOUT ``--run-id`` → rc 1 structured refusal. ``run_project``'s step-0 fail-fast (no wall-clock default) surfaces through the CLI refusal wrapper, no traceback.""" outbox = tmp_path / "out" outbox.mkdir() rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--outbox-dir", str(outbox), "--live-dry-run", ] ) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_dimension_config_missing_file_refuses(capsys) -> None: """(d) ``--dimension-config `` → rc 1 structured refusal. ``load_dimension`` raises ``FileNotFoundError``, caught by the WIDENED dry-run handler (not just ``ValueError`` — Pass-2 #1).""" rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--dimension-config", "/nonexistent-dim-config.json", "--live-dry-run", ] ) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() # --- Step 4: mode-exclusivity refusals + backward-compat pin ------------------------------------- def _met_portfolio_goal(tmp_path: Path) -> tuple[Path, Path]: """A portfolio-hard goal (1 øre) already met by a 1-øre ledger — so any RED (pre-refusal) fall-through into the portfolio dispatch stops OFFLINE at the goal check (no client, no socket).""" goals = tmp_path / "goals.json" goals.write_text('{"portfolio": {"absolute_ore": 1, "mode": "hard"}}', encoding="utf-8") led = SavingsLedger() led.add_realized( LedgerEntry( project_id="FV42-GSV-E1", dimension="energi", candidate_identity="c1", amount_ore=1, verdict_id="v1", provenance="x", ) ) ledger = tmp_path / "ledger.json" led.save(str(ledger)) return goals, ledger def test_goals_without_portfolio_refuses(tmp_path, capsys) -> None: """(a) ``--goals`` without ``--portfolio`` → rc 1 refusal (the flag belongs to portfolio mode). ``--live-dry-run`` keeps the RED (pre-refusal) fall-through offline.""" goals = tmp_path / "goals.json" goals.write_text('{"portfolio": {"absolute_ore": 1, "mode": "hard"}}', encoding="utf-8") rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--goals", str(goals), "--live-dry-run", ] ) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_ledger_without_portfolio_refuses(tmp_path, capsys) -> None: """(a') ``--ledger`` without ``--portfolio`` → rc 1 refusal.""" ledger = tmp_path / "ledger.json" ledger.write_text("[]", encoding="utf-8") rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--ledger", str(ledger), "--live-dry-run", ] ) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_portfolio_with_single_project_flag_refuses(tmp_path, capsys) -> None: """(c) ``--portfolio`` combined with a single-project-only flag (``--outbox-dir``) → rc 1 refusal naming the offending flag. The met portfolio goal keeps the RED fall-through offline.""" goals, ledger = _met_portfolio_goal(tmp_path) rc = run.main( [ "--portfolio", "--goals", str(goals), "--ledger", str(ledger), "--outbox-dir", str(tmp_path / "ob"), ] ) assert rc == 1 err = capsys.readouterr().err.lower() assert "refused" in err assert "--outbox-dir" in err def test_single_project_mode_without_docs_dir_refuses(capsys) -> None: """(b) single-project mode with a pid but no ``--docs-dir`` → rc 1 refusal (the Step-3 compensating guard for the relaxed argparse ``required=``).""" rc = run.main([_PID]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_single_project_mode_without_pid_refuses(capsys) -> None: """(b') single-project mode with ``--docs-dir`` but no PROJECT_ID → rc 1 refusal.""" rc = run.main(["--docs-dir", str(BUNDLE_DIR)]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_legacy_single_project_invocation_still_succeeds(capsys) -> None: """Backward-compat pin: the legacy invocation (positional pid + ``--docs-dir`` + ``--bundle-dir`` + ``--live-dry-run``) still returns rc 0 — the existing CLI contract survives Step 3's ``nargs='?'``/``required`` relaxation.""" rc = run.main( [_PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--live-dry-run"] ) assert rc == 0 assert "LIVE-DRY-RUN OK" in capsys.readouterr().out # --- Step 5: confirm coverage for the already-wired flags (never re-wired; run.py untouched) ------ def test_verdict_dir_ingested_at_main_level_offline(tmp_path, capsys) -> None: """Step 5 (SC2 second half): ``--verdict-dir`` is exercised at ``main()`` level — the previously untested already-wired flag. The async inbox is ingested (``load_verdicts_from_dir``, ``run.py:287``) BEFORE the dry-run cut (``run.py:335``), so a dropped verdict is threaded through ``main()`` offline without raising. ``--bundle-dir``'s ``main()``-level coverage already exists in ``tests/test_live_dry_run.py:32-49`` and is NOT re-tested here (never re-wired).""" inbox = tmp_path / "inbox" feats = ProposalFeatures( affected_codes=frozenset({"ENERGI-TOTAL-EL"}), measure_type="energy_efficiency", claimed_saving_nok=30000.0, description="LED-retrofit", ) write_verdict(str(inbox), capture_verdict(feats, "approved", "expert reviewed (sim)")) rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--verdict-dir", str(inbox), "--live-dry-run", ] ) assert rc == 0 assert "LIVE-DRY-RUN OK" in capsys.readouterr().out # --- S5.4: --report / --json read-only value-report mode ------------------------------------------ def _report_ledger(tmp_path: Path) -> Path: """A saved ledger with >=2 projects and one cross-dimension overlap (``c-a`` under both ``energi`` and ``asfalt`` in FV42 -> counted once, flagged), for the value-report arms. portfolio_total = 1234567 (overlap once) + 500000 = 1734567 øre.""" led = SavingsLedger() led.add_realized( LedgerEntry( project_id="FV42-GSV-E1", dimension="energi", candidate_identity="c-a", amount_ore=1234567, verdict_id="v1", provenance="p1", ) ) led.add_realized( LedgerEntry( project_id="FV42-GSV-E1", dimension="asfalt", candidate_identity="c-a", amount_ore=1234567, verdict_id="v2", provenance="p2", # cross-dimension overlap on (FV42-GSV-E1, c-a) ) ) led.add_realized( LedgerEntry( project_id="RV13-RAS-TP", dimension="energi", candidate_identity="c-b", amount_ore=500000, verdict_id="v3", provenance="p3", ) ) p = tmp_path / "ledger.json" led.save(str(p)) return p def test_report_prints_table_rc0(tmp_path, capsys) -> None: """SC3: ``--report --ledger `` -> rc 0; stdout carries a per-project row + the portfolio-total NOK string. Dispatched FIRST, so no PROJECT_ID/--docs-dir is needed (no single-project refusal).""" rc = run.main(["--report", "--ledger", str(_report_ledger(tmp_path))]) out = capsys.readouterr().out assert rc == 0 assert "FV42-GSV-E1" in out # a per-project row assert "17\xa0345,67\xa0kr" in out # portfolio total 1734567 øre def test_report_json_rc0_parses_rollup(tmp_path, capsys) -> None: """SC4: ``--report ... --json`` -> rc 0 and ``json.loads(stdout)`` yields the roll-up (int portfolio total, per_project dict, overlaps as JSON lists, provenance list).""" rc = run.main(["--report", "--ledger", str(_report_ledger(tmp_path)), "--json"]) out = capsys.readouterr().out assert rc == 0 payload = json.loads(out) assert payload["portfolio_total_ore"] == 1734567 assert payload["per_project"]["FV42-GSV-E1"] == 1234567 assert isinstance(payload["overlaps"], list) assert ["FV42-GSV-E1", "c-a"] in payload["overlaps"] # tuple serialized as a JSON array assert isinstance(payload["provenance"], list) assert len(payload["provenance"]) == 3 # one ProvenanceLine per ledger entry def test_report_missing_ledger_file_rc1(capsys) -> None: """SC5: ``--report --ledger /nonexistent`` -> rc 1, stderr non-empty, NO table on stdout (a load failure must never masquerade as a real zero-savings result).""" rc = run.main(["--report", "--ledger", "/nonexistent-ledger.json"]) cap = capsys.readouterr() assert rc == 1 assert cap.err.strip() assert cap.out == "" def test_report_malformed_ledger_rc1(tmp_path, capsys) -> None: """SC5: a malformed-row ledger file -> rc 1 (``ValidationError`` surfaced as a refusal).""" bad = tmp_path / "bad.json" bad.write_text('[{"project_id": "P1"}]', encoding="utf-8") # missing required fields rc = run.main(["--report", "--ledger", str(bad)]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_report_empty_dict_ledger_rc1(tmp_path, capsys) -> None: """SC5 masquerade guard: a valid-JSON ``{}`` ledger must NOT load as an empty ledger and print a misleading ``0,00 kr`` at rc 0 — a malformed file masquerading as a real zero-savings result is the exact failure SC5's fail-fast refusal exists to prevent.""" bad = tmp_path / "empty-obj.json" bad.write_text("{}", encoding="utf-8") rc = run.main(["--report", "--ledger", str(bad)]) cap = capsys.readouterr() assert rc == 1 assert "refused" in cap.err.lower() assert cap.out == "" # no table, no "0,00 kr" def test_report_nonlist_ledger_rc1_no_traceback(tmp_path, capsys) -> None: """SC5: a valid-JSON but wrong-shape ledger (bare scalar / object-with-keys) -> rc 1 refusal, NOT an uncaught ``TypeError`` traceback ('rc 1, no traceback').""" bad = tmp_path / "scalar.json" bad.write_text("42", encoding="utf-8") rc = run.main(["--report", "--ledger", str(bad)]) cap = capsys.readouterr() assert rc == 1 assert "refused" in cap.err.lower() assert "Traceback" not in cap.err assert cap.out == "" def test_report_without_ledger_rc1_no_traceback(capsys) -> None: """Major-#1 guard: ``--report`` with no ``--ledger`` -> rc 1 with a 'requires --ledger' message and no traceback (guards ``SavingsLedger.load(None)`` -> ``Path(None)`` TypeError).""" rc = run.main(["--report"]) err = capsys.readouterr().err assert rc == 1 assert "requires --ledger" in err assert "Traceback" not in err def test_report_with_portfolio_refuses(tmp_path, capsys) -> None: """Major-#3 partition: ``--report`` + ``--portfolio`` -> rc 1 (mode-exclusive).""" rc = run.main(["--report", "--portfolio", "--ledger", str(_report_ledger(tmp_path))]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_report_with_goals_refuses(tmp_path, capsys) -> None: """Major-#3 / P2-2 allowlist: ``--report`` + ``--goals`` (a config flag) -> rc 1. The allowlist rejects config flags too, not just the two mode flags — else ``--goals`` would be silently dropped, whereas bare ``--goals`` is refused (adding ``--report`` must not suppress a refusal).""" goals = tmp_path / "goals.json" goals.write_text('{"portfolio": {"absolute_ore": 1, "mode": "hard"}}', encoding="utf-8") rc = run.main(["--report", "--goals", str(goals), "--ledger", str(_report_ledger(tmp_path))]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() def test_json_without_report_refuses(tmp_path, capsys) -> None: """Major-#3 partition: ``--json`` without ``--report`` -> rc 1 (a stray --json is never silently ignored).""" rc = run.main(["--json", "--ledger", str(_report_ledger(tmp_path))]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() # --- S3.1 SC6: --semantic-retrieval opt-in (default OFF) -------------------------------------- _ENERGY_REPLY = ( '{"measure":"LED-retrofit av kontorbelysning","affected_items":' '[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}' ) _VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"} # The marker lives in the RATIONALE, which is what ``format_fewshot`` emits into the prompt. # "0.37" appears nowhere in the bundle (grep-verified), so its presence in a prompt can only have # come through the ExpeL fold — it cannot be leaked by bundle context. _TIE_MARKER = "realiseringsgrad=0.37" _MARKER_ID = "zz-s31-run-marker" _DISTRACTOR_ID = "aa-s31-run-distractor" def _tied_pair_store(): """Two verdicts that are structurally IDENTICAL to the bundle's candidate (similarity 1.0 each), so the structural ranking can only separate them on ``id`` — and ``_MARKER_ID`` loses that tie. Only the cosine term can promote it. The bundle's own seed verdict is deliberately NOT included: its features are byte-identical to the query, making it a perfect cosine match that would win every hybrid ranking and so could never demonstrate a tie-break.""" from portfolio_optimiser.verdicts import Verdict, VerdictStore, bundle_candidate_features query = bundle_candidate_features(str(BUNDLE_DIR)) def tied(verdict_id: str, description: str, rationale: str) -> Verdict: return Verdict( id=verdict_id, proposal_features=ProposalFeatures( affected_codes=query.affected_codes, measure_type=query.measure_type, claimed_saving_nok=query.claimed_saving_nok, description=description, ), decision="approved", rationale=rationale, ) return VerdictStore( verdicts=[ tied( _DISTRACTOR_ID, "avvist forslag om reforhandling av renholdskontrakt i administrasjonsbygget", "ingen realiseringsdata for dette tiltaket", ), tied( _MARKER_ID, "LED-retrofit i kontorlokaler: 90 W armaturer erstattet med 40 W", f"tidligere LED-dom [{_TIE_MARKER}]", ), ] ) def _generation_prompts(sink: list[str]) -> list[str]: return [p for p in sink if "SavingsProposal" in p] def test_semantic_retrieval_flag_parses_offline(capsys) -> None: """(a) the flag is accepted in single-project mode and the run still stops offline.""" rc = run.main( [ _PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--semantic-retrieval", "--live-dry-run", ] ) assert rc == 0 assert "LIVE-DRY-RUN OK" in capsys.readouterr().out def test_semantic_retrieval_is_not_refused_in_portfolio_mode(capsys) -> None: """The flag is valid in BOTH modes (like --dimension-config), so the portfolio partition must not name it. Probed via a run that IS refused for a different flag: the refusal lists ``--docs-dir`` and must NOT mention ``--semantic-retrieval``.""" rc = run.main(["--portfolio", "--semantic-retrieval", "--docs-dir", str(BUNDLE_DIR)]) err = capsys.readouterr().err assert rc == 1 assert "--docs-dir" in err assert "--semantic-retrieval" not in err def test_report_with_semantic_retrieval_is_refused(tmp_path, capsys) -> None: """--report is an ALLOWLIST: only --ledger/--json ride along. A silently-dropped --semantic-retrieval would break that partition.""" ledger_file = tmp_path / "ledger.json" SavingsLedger(entries=[]).save(str(ledger_file)) rc = run.main(["--report", "--ledger", str(ledger_file), "--semantic-retrieval"]) assert rc == 1 assert "refused" in capsys.readouterr().err.lower() async def test_semantic_retrieval_on_swaps_the_fewshot_reaching_the_prompt( make_recording_client_factory, ) -> None: """(c) the substantive run-level proof: with ``semantic_retrieval=True`` and ``top_k=1``, the verdict that only cosine can surface is the one whose rationale reaches the hypothesis prompt. Drives the REAL Step-1 fold via the recording client — not ``--live-dry-run``, which returns before the fold.""" factory, recorded = make_recording_client_factory(_ENERGY_REPLY) await run_project( _PID, "local", docs_dir=str(BUNDLE_DIR), bundle_dir=str(BUNDLE_DIR), verdict_input=_VERDICT_INPUT, store=_tied_pair_store(), client_factory=factory, top_k=1, semantic_retrieval=True, ) gen_prompts = _generation_prompts(recorded) assert gen_prompts, "the generation call must have happened" assert any(_TIE_MARKER in p for p in gen_prompts), ( "the cosine-surfaced verdict did not reach the hypothesis prompt — " "--semantic-retrieval is not installing the HybridRanker before the Step-1 fold" ) assert any(_MARKER_ID in p for p in gen_prompts) async def test_semantic_retrieval_off_leaves_the_structural_pick_in_the_prompt( make_recording_client_factory, ) -> None: """CAUSALITY CONTROL — the identical run with the flag OFF must carry the STRUCTURAL winner instead, and no marker. This is what makes the positive above load-bearing: the swap is caused by the flag, not by the fixture merely containing the marker.""" factory, recorded = make_recording_client_factory(_ENERGY_REPLY) await run_project( _PID, "local", docs_dir=str(BUNDLE_DIR), bundle_dir=str(BUNDLE_DIR), verdict_input=_VERDICT_INPUT, store=_tied_pair_store(), client_factory=factory, top_k=1, ) assert all(_TIE_MARKER not in p for p in recorded), ( "the marker reached a prompt with --semantic-retrieval OFF — the default path is not the " "structural ranking, or the assertion is not load-bearing" ) gen_prompts = _generation_prompts(recorded) assert any(_DISTRACTOR_ID in p for p in gen_prompts), ( "the structural winner did not reach the prompt — the default fold is broken" )