portfolio-optimiser/tests/test_run_cli.py

724 lines
27 KiB
Python

"""S5.3 CLI-parity tests for ``run.main()`` single-project flags (Steps 2/4/5).
In-process ``run.main([argv])`` + rc + ``capsys`` substring asserts (never subprocess),
mirroring ``tests/test_live_dry_run.py``. Every arm is offline — it stops before the first
model call (``debate.run``), so no socket/network is exercised (brief NFR). The bundle
fixture ``shared/examples/bygg-energi-mikro`` (project ``BYGG-KONTOR-NORD``) supplies citable
content so the dry-run reaches its offline return.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from portfolio_optimiser import run
from portfolio_optimiser.dimension import Dimension
from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger
from portfolio_optimiser.run import run_project
from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict
BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
_PID = "BYGG-KONTOR-NORD"
@pytest.fixture(autouse=True)
def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None:
"""Hermetic env (verbatim from ``test_live_dry_run.py``): clear the S4.1 out-of-tree overrides
so these CLI assertions read the BUNDLED map/config, not the operator's Foundry environment."""
monkeypatch.delenv("PORTFOLIO_MODEL_MAP", raising=False)
monkeypatch.delenv("PORTFOLIO_FOUNDRY_PROJECT_ENDPOINT", raising=False)
def _write_dimension(tmp_path: Path) -> Path:
"""A valid dimension scope config on disk (loaded fail-fast by ``load_dimension``)."""
dim = Dimension(
id="energi",
label="Energi",
allowed_measure_types=frozenset({"energy_efficiency"}),
)
p = tmp_path / "dim.json"
p.write_text(dim.model_dump_json(), encoding="utf-8")
return p
# --- Step 2: --dimension-config / --outbox-dir / --run-id wiring + structured refusal -------------
def test_dimension_config_flag_parses_offline(tmp_path, capsys) -> None:
"""(a) ``--dimension-config <valid>`` + ``--live-dry-run`` → rc 0 (flag parsed, loader invoked,
offline — stops before any model call)."""
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--dimension-config",
str(_write_dimension(tmp_path)),
"--live-dry-run",
]
)
assert rc == 0
assert "LIVE-DRY-RUN OK" in capsys.readouterr().out
def test_outbox_dir_with_run_id_writes_runconfig_offline(tmp_path) -> None:
"""(b) ``--outbox-dir`` + ``--run-id`` + ``--live-dry-run`` → rc 0 AND ``<tmp>/r1-runconfig.json``
written offline (via ``write_run_config``, before the dry-run return)."""
outbox = tmp_path / "out"
outbox.mkdir()
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--outbox-dir",
str(outbox),
"--run-id",
"r1",
"--live-dry-run",
]
)
assert rc == 0
assert (outbox / "r1-runconfig.json").is_file()
def test_outbox_dir_without_run_id_refuses(tmp_path, capsys) -> None:
"""(c) RED guard: ``--outbox-dir`` WITHOUT ``--run-id`` → rc 1 structured refusal. ``run_project``'s
step-0 fail-fast (no wall-clock default) surfaces through the CLI refusal wrapper, no traceback."""
outbox = tmp_path / "out"
outbox.mkdir()
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--outbox-dir",
str(outbox),
"--live-dry-run",
]
)
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_dimension_config_missing_file_refuses(capsys) -> None:
"""(d) ``--dimension-config <nonexistent>`` → rc 1 structured refusal. ``load_dimension`` raises
``FileNotFoundError``, caught by the WIDENED dry-run handler (not just ``ValueError`` — Pass-2 #1)."""
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--dimension-config",
"/nonexistent-dim-config.json",
"--live-dry-run",
]
)
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
# --- Step 4: mode-exclusivity refusals + backward-compat pin -------------------------------------
def _met_portfolio_goal(tmp_path: Path) -> tuple[Path, Path]:
"""A portfolio-hard goal (1 øre) already met by a 1-øre ledger — so any RED (pre-refusal)
fall-through into the portfolio dispatch stops OFFLINE at the goal check (no client, no socket)."""
goals = tmp_path / "goals.json"
goals.write_text('{"portfolio": {"absolute_ore": 1, "mode": "hard"}}', encoding="utf-8")
led = SavingsLedger()
led.add_realized(
LedgerEntry(
project_id="FV42-GSV-E1",
dimension="energi",
candidate_identity="c1",
amount_ore=1,
verdict_id="v1",
provenance="x",
)
)
ledger = tmp_path / "ledger.json"
led.save(str(ledger))
return goals, ledger
def test_goals_without_portfolio_refuses(tmp_path, capsys) -> None:
"""(a) ``--goals`` without ``--portfolio`` → rc 1 refusal (the flag belongs to portfolio mode).
``--live-dry-run`` keeps the RED (pre-refusal) fall-through offline."""
goals = tmp_path / "goals.json"
goals.write_text('{"portfolio": {"absolute_ore": 1, "mode": "hard"}}', encoding="utf-8")
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--goals",
str(goals),
"--live-dry-run",
]
)
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_ledger_without_portfolio_refuses(tmp_path, capsys) -> None:
"""(a') ``--ledger`` without ``--portfolio`` → rc 1 refusal."""
ledger = tmp_path / "ledger.json"
ledger.write_text("[]", encoding="utf-8")
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--ledger",
str(ledger),
"--live-dry-run",
]
)
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_portfolio_with_single_project_flag_refuses(tmp_path, capsys) -> None:
"""(c) ``--portfolio`` combined with a single-project-only flag (``--outbox-dir``) → rc 1 refusal
naming the offending flag. The met portfolio goal keeps the RED fall-through offline."""
goals, ledger = _met_portfolio_goal(tmp_path)
rc = run.main(
[
"--portfolio",
"--goals",
str(goals),
"--ledger",
str(ledger),
"--outbox-dir",
str(tmp_path / "ob"),
]
)
assert rc == 1
err = capsys.readouterr().err.lower()
assert "refused" in err
assert "--outbox-dir" in err
def test_single_project_mode_without_docs_dir_refuses(capsys) -> None:
"""(b) single-project mode with a pid but no ``--docs-dir`` → rc 1 refusal (the Step-3
compensating guard for the relaxed argparse ``required=``)."""
rc = run.main([_PID])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_single_project_mode_without_pid_refuses(capsys) -> None:
"""(b') single-project mode with ``--docs-dir`` but no PROJECT_ID → rc 1 refusal."""
rc = run.main(["--docs-dir", str(BUNDLE_DIR)])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_legacy_single_project_invocation_still_succeeds(capsys) -> None:
"""Backward-compat pin: the legacy invocation (positional pid + ``--docs-dir`` + ``--bundle-dir``
+ ``--live-dry-run``) still returns rc 0 — the existing CLI contract survives Step 3's
``nargs='?'``/``required`` relaxation."""
rc = run.main(
[_PID, "--docs-dir", str(BUNDLE_DIR), "--bundle-dir", str(BUNDLE_DIR), "--live-dry-run"]
)
assert rc == 0
assert "LIVE-DRY-RUN OK" in capsys.readouterr().out
# --- Step 5: confirm coverage for the already-wired flags (never re-wired; run.py untouched) ------
def test_verdict_dir_ingested_at_main_level_offline(tmp_path, capsys) -> None:
"""Step 5 (SC2 second half): ``--verdict-dir`` is exercised at ``main()`` level — the previously
untested already-wired flag. The async inbox is ingested (``load_verdicts_from_dir``,
``run.py:287``) BEFORE the dry-run cut (``run.py:335``), so a dropped verdict is threaded through
``main()`` offline without raising. ``--bundle-dir``'s ``main()``-level coverage already exists
in ``tests/test_live_dry_run.py:32-49`` and is NOT re-tested here (never re-wired)."""
inbox = tmp_path / "inbox"
feats = ProposalFeatures(
affected_codes=frozenset({"ENERGI-TOTAL-EL"}),
measure_type="energy_efficiency",
claimed_saving_nok=30000.0,
description="LED-retrofit",
)
write_verdict(str(inbox), capture_verdict(feats, "approved", "expert reviewed (sim)"))
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--verdict-dir",
str(inbox),
"--live-dry-run",
]
)
assert rc == 0
assert "LIVE-DRY-RUN OK" in capsys.readouterr().out
# --- S5.4: --report / --json read-only value-report mode ------------------------------------------
def _report_ledger(tmp_path: Path) -> Path:
"""A saved ledger with >=2 projects and one cross-dimension overlap (``c-a`` under both
``energi`` and ``asfalt`` in FV42 -> counted once, flagged), for the value-report arms.
portfolio_total = 1234567 (overlap once) + 500000 = 1734567 øre."""
led = SavingsLedger()
led.add_realized(
LedgerEntry(
project_id="FV42-GSV-E1",
dimension="energi",
candidate_identity="c-a",
amount_ore=1234567,
verdict_id="v1",
provenance="p1",
)
)
led.add_realized(
LedgerEntry(
project_id="FV42-GSV-E1",
dimension="asfalt",
candidate_identity="c-a",
amount_ore=1234567,
verdict_id="v2",
provenance="p2", # cross-dimension overlap on (FV42-GSV-E1, c-a)
)
)
led.add_realized(
LedgerEntry(
project_id="RV13-RAS-TP",
dimension="energi",
candidate_identity="c-b",
amount_ore=500000,
verdict_id="v3",
provenance="p3",
)
)
p = tmp_path / "ledger.json"
led.save(str(p))
return p
def test_report_prints_table_rc0(tmp_path, capsys) -> None:
"""SC3: ``--report --ledger <f>`` -> rc 0; stdout carries a per-project row + the portfolio-total
NOK string. Dispatched FIRST, so no PROJECT_ID/--docs-dir is needed (no single-project refusal)."""
rc = run.main(["--report", "--ledger", str(_report_ledger(tmp_path))])
out = capsys.readouterr().out
assert rc == 0
assert "FV42-GSV-E1" in out # a per-project row
assert "17\xa0345,67\xa0kr" in out # portfolio total 1734567 øre
def test_report_json_rc0_parses_rollup(tmp_path, capsys) -> None:
"""SC4: ``--report ... --json`` -> rc 0 and ``json.loads(stdout)`` yields the roll-up
(int portfolio total, per_project dict, overlaps as JSON lists, provenance list)."""
rc = run.main(["--report", "--ledger", str(_report_ledger(tmp_path)), "--json"])
out = capsys.readouterr().out
assert rc == 0
payload = json.loads(out)
assert payload["portfolio_total_ore"] == 1734567
assert payload["per_project"]["FV42-GSV-E1"] == 1234567
assert isinstance(payload["overlaps"], list)
assert ["FV42-GSV-E1", "c-a"] in payload["overlaps"] # tuple serialized as a JSON array
assert isinstance(payload["provenance"], list)
assert len(payload["provenance"]) == 3 # one ProvenanceLine per ledger entry
def test_report_missing_ledger_file_rc1(capsys) -> None:
"""SC5: ``--report --ledger /nonexistent`` -> rc 1, stderr non-empty, NO table on stdout (a load
failure must never masquerade as a real zero-savings result)."""
rc = run.main(["--report", "--ledger", "/nonexistent-ledger.json"])
cap = capsys.readouterr()
assert rc == 1
assert cap.err.strip()
assert cap.out == ""
def test_report_malformed_ledger_rc1(tmp_path, capsys) -> None:
"""SC5: a malformed-row ledger file -> rc 1 (``ValidationError`` surfaced as a refusal)."""
bad = tmp_path / "bad.json"
bad.write_text('[{"project_id": "P1"}]', encoding="utf-8") # missing required fields
rc = run.main(["--report", "--ledger", str(bad)])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_report_empty_dict_ledger_rc1(tmp_path, capsys) -> None:
"""SC5 masquerade guard: a valid-JSON ``{}`` ledger must NOT load as an empty ledger and print a
misleading ``0,00 kr`` at rc 0 — a malformed file masquerading as a real zero-savings result is
the exact failure SC5's fail-fast refusal exists to prevent."""
bad = tmp_path / "empty-obj.json"
bad.write_text("{}", encoding="utf-8")
rc = run.main(["--report", "--ledger", str(bad)])
cap = capsys.readouterr()
assert rc == 1
assert "refused" in cap.err.lower()
assert cap.out == "" # no table, no "0,00 kr"
def test_report_nonlist_ledger_rc1_no_traceback(tmp_path, capsys) -> None:
"""SC5: a valid-JSON but wrong-shape ledger (bare scalar / object-with-keys) -> rc 1 refusal, NOT
an uncaught ``TypeError`` traceback ('rc 1, no traceback')."""
bad = tmp_path / "scalar.json"
bad.write_text("42", encoding="utf-8")
rc = run.main(["--report", "--ledger", str(bad)])
cap = capsys.readouterr()
assert rc == 1
assert "refused" in cap.err.lower()
assert "Traceback" not in cap.err
assert cap.out == ""
def test_report_without_ledger_rc1_no_traceback(capsys) -> None:
"""Major-#1 guard: ``--report`` with no ``--ledger`` -> rc 1 with a 'requires --ledger' message
and no traceback (guards ``SavingsLedger.load(None)`` -> ``Path(None)`` TypeError)."""
rc = run.main(["--report"])
err = capsys.readouterr().err
assert rc == 1
assert "requires --ledger" in err
assert "Traceback" not in err
def test_report_with_portfolio_refuses(tmp_path, capsys) -> None:
"""Major-#3 partition: ``--report`` + ``--portfolio`` -> rc 1 (mode-exclusive)."""
rc = run.main(["--report", "--portfolio", "--ledger", str(_report_ledger(tmp_path))])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_report_with_goals_refuses(tmp_path, capsys) -> None:
"""Major-#3 / P2-2 allowlist: ``--report`` + ``--goals`` (a config flag) -> rc 1. The allowlist
rejects config flags too, not just the two mode flags — else ``--goals`` would be silently
dropped, whereas bare ``--goals`` is refused (adding ``--report`` must not suppress a refusal)."""
goals = tmp_path / "goals.json"
goals.write_text('{"portfolio": {"absolute_ore": 1, "mode": "hard"}}', encoding="utf-8")
rc = run.main(["--report", "--goals", str(goals), "--ledger", str(_report_ledger(tmp_path))])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
def test_json_without_report_refuses(tmp_path, capsys) -> None:
"""Major-#3 partition: ``--json`` without ``--report`` -> rc 1 (a stray --json is never silently
ignored)."""
rc = run.main(["--json", "--ledger", str(_report_ledger(tmp_path))])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
# --- S3.1 SC6: --semantic-retrieval opt-in (default OFF) --------------------------------------
_ENERGY_REPLY = (
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
)
_VERDICT_INPUT = {"decision": "approved", "rationale": "expert reviewed (sim)"}
# The marker lives in the RATIONALE, which is what ``format_fewshot`` emits into the prompt.
# "0.37" appears nowhere in the bundle (grep-verified), so its presence in a prompt can only have
# come through the ExpeL fold — it cannot be leaked by bundle context.
_TIE_MARKER = "realiseringsgrad=0.37"
# The pair ties on the STRUCTURAL score while carrying different code sets: each keeps the query's
# own code and adds one foreign code, so Jaccard is equal (0.700 each) but the canonical embedding
# string differs and cosine has something to separate. The previous fixture tied by being
# structurally IDENTICAL and differing only in prose — a shape the framework cannot mint, since
# ``_mint_id`` ignores description and both candidates would collapse onto one id.
_MARKER_EXTRA_CODE = "01.1"
_DISTRACTOR_EXTRA_CODE = "01.4"
def _tied_pair_verdicts():
"""The tied pair, minted through ``capture_verdict`` — the real minting path.
The bundle's own seed verdict is deliberately NOT included: its features are identical to the
query, making it a perfect cosine match that would win every hybrid ranking and so could never
demonstrate a tie-break."""
from portfolio_optimiser.verdicts import (
VerdictStore,
bundle_candidate_features,
capture_verdict,
)
query = bundle_candidate_features(str(BUNDLE_DIR))
def tied(extra_code: str, decision: str, rationale: str):
return capture_verdict(
ProposalFeatures(
affected_codes=query.affected_codes | {extra_code},
measure_type=query.measure_type,
# ``description == measure`` is what both live minting paths emit
# (``run._features_of`` / ``verdicts._features_from_ir``).
claimed_saving_nok=query.claimed_saving_nok,
description=query.measure_type,
),
decision,
rationale,
)
distractor = tied(
_DISTRACTOR_EXTRA_CODE,
"rejected",
"ingen realiseringsdata for dette tiltaket",
)
marker = tied(_MARKER_EXTRA_CODE, "approved", f"tidligere LED-dom [{_TIE_MARKER}]")
return VerdictStore(verdicts=[distractor, marker]), marker.id, distractor.id
# Minted once at module scope so the tests can name the ids. ``_MARKER_ID`` sorts AFTER
# ``_DISTRACTOR_ID``, so the structural ``(-similarity, id)`` key puts the DISTRACTOR first and
# only the cosine term can overturn it.
_MARKER_ID = _tied_pair_verdicts()[1]
_DISTRACTOR_ID = _tied_pair_verdicts()[2]
def _tied_pair_store():
return _tied_pair_verdicts()[0]
def _bundle_reference_project(tmp_path: Path):
"""A single bundle-backed project for the portfolio arm, pointed at the SAME bundle the tied
pair was derived from — a different bundle would yield a different query and untie the pair.
Topology mirrors ``tests/test_portfolio_learning_loadbearing.py::_bundle_kplus1``."""
from portfolio_optimiser.reference_domain import Project
docs = tmp_path / "portfolio-docs"
docs.mkdir()
(docs / "cost.txt").write_text(
"LED-retrofit av lysrorarmaturer i kontorlokaler reduserte energikostnaden.",
encoding="utf-8",
)
return Project(
id=_PID,
name="Bundle-backed portfolio project",
description="bundle-backed project for the semantic-retrieval forwarding proof",
currency="NOK",
cost_items=(),
docs_dir=str(docs),
verdict_input=_VERDICT_INPUT,
bundle_dir=str(BUNDLE_DIR),
verdict_dir=None,
)
def _generation_prompts(sink: list[str]) -> list[str]:
return [p for p in sink if "SavingsProposal" in p]
def test_semantic_retrieval_flag_parses_offline(capsys) -> None:
"""(a) the flag is accepted in single-project mode and the run still stops offline."""
rc = run.main(
[
_PID,
"--docs-dir",
str(BUNDLE_DIR),
"--bundle-dir",
str(BUNDLE_DIR),
"--semantic-retrieval",
"--live-dry-run",
]
)
assert rc == 0
assert "LIVE-DRY-RUN OK" in capsys.readouterr().out
def test_semantic_retrieval_is_not_refused_in_portfolio_mode(capsys) -> None:
"""The flag is valid in BOTH modes (like --dimension-config), so the portfolio partition must
not name it. Probed via a run that IS refused for a different flag: the refusal lists
``--docs-dir`` and must NOT mention ``--semantic-retrieval``."""
rc = run.main(["--portfolio", "--semantic-retrieval", "--docs-dir", str(BUNDLE_DIR)])
err = capsys.readouterr().err
assert rc == 1
assert "--docs-dir" in err
assert "--semantic-retrieval" not in err
def test_report_with_semantic_retrieval_is_refused(tmp_path, capsys) -> None:
"""--report is an ALLOWLIST: only --ledger/--json ride along. A silently-dropped
--semantic-retrieval would break that partition."""
ledger_file = tmp_path / "ledger.json"
SavingsLedger(entries=[]).save(str(ledger_file))
rc = run.main(["--report", "--ledger", str(ledger_file), "--semantic-retrieval"])
assert rc == 1
assert "refused" in capsys.readouterr().err.lower()
async def test_semantic_retrieval_on_swaps_the_fewshot_reaching_the_prompt(
make_recording_client_factory,
) -> None:
"""(c) the substantive run-level proof: with ``semantic_retrieval=True`` and ``top_k=1``, the
verdict that only cosine can surface is the one whose rationale reaches the hypothesis prompt.
Drives the REAL Step-1 fold via the recording client — not ``--live-dry-run``, which returns
before the fold."""
factory, recorded = make_recording_client_factory(_ENERGY_REPLY)
await run_project(
_PID,
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=_tied_pair_store(),
client_factory=factory,
top_k=1,
semantic_retrieval=True,
)
gen_prompts = _generation_prompts(recorded)
assert gen_prompts, "the generation call must have happened"
assert any(_TIE_MARKER in p for p in gen_prompts), (
"the cosine-surfaced verdict did not reach the hypothesis prompt — "
"--semantic-retrieval is not installing the HybridRanker before the Step-1 fold"
)
assert any(_MARKER_ID in p for p in gen_prompts)
async def test_semantic_retrieval_off_leaves_the_structural_pick_in_the_prompt(
make_recording_client_factory,
) -> None:
"""CAUSALITY CONTROL — the identical run with the flag OFF must carry the STRUCTURAL winner
instead, and no marker. This is what makes the positive above load-bearing: the swap is caused
by the flag, not by the fixture merely containing the marker."""
factory, recorded = make_recording_client_factory(_ENERGY_REPLY)
await run_project(
_PID,
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=_tied_pair_store(),
client_factory=factory,
top_k=1,
)
assert all(_TIE_MARKER not in p for p in recorded), (
"the marker reached a prompt with --semantic-retrieval OFF — the default path is not the "
"structural ranking, or the assertion is not load-bearing"
)
gen_prompts = _generation_prompts(recorded)
assert any(_DISTRACTOR_ID in p for p in gen_prompts), (
"the structural winner did not reach the prompt — the default fold is broken"
)
async def test_semantic_retrieval_does_not_leak_into_a_reused_store(
make_recording_client_factory,
) -> None:
"""LEAK CONTROL — the opt-in must not outlive the run that asked for it.
The store is caller-owned. Installing the ranker on it (``store.retriever = ...``) meant a
flag-ON run silently governed every LATER retrieval on that same object, so a subsequent run
with the flag OFF still ranked semantically. Here the SAME store instance is driven twice:
once with the flag on, once without. The second run must give the STRUCTURAL pick, and the
store must be left exactly as the caller handed it over.
Detach point: assign ``store.retriever`` in ``run_project`` instead of passing the ranker per
call → RED."""
store = _tied_pair_store()
assert store.retriever is None # precondition: the caller handed over a clean store
factory_on, _ = make_recording_client_factory(_ENERGY_REPLY)
await run_project(
_PID,
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=store,
client_factory=factory_on,
top_k=1,
semantic_retrieval=True,
)
factory_off, recorded_off = make_recording_client_factory(_ENERGY_REPLY)
await run_project(
_PID,
"local",
docs_dir=str(BUNDLE_DIR),
bundle_dir=str(BUNDLE_DIR),
verdict_input=_VERDICT_INPUT,
store=store,
client_factory=factory_off,
top_k=1,
)
assert store.retriever is None, (
"run_project mutated the caller's store — the opt-in leaked out of the run that asked "
"for it"
)
assert all(_TIE_MARKER not in p for p in _generation_prompts(recorded_off)), (
"the flag-OFF run still ranked semantically — the previous run's opt-in leaked through "
"the shared store"
)
async def test_run_portfolio_forwards_semantic_retrieval_to_each_project(
make_recording_client_factory, monkeypatch, tmp_path
) -> None:
"""WIRING — ``run_portfolio(semantic_retrieval=True)`` must reach each project's Step-1 fold.
No test in the repo drove the portfolio path with this flag before: the pre-loop
``store.retriever`` install was covered only by the single-project tests, so deleting the
forwarding would have gone unnoticed.
Detach point: drop ``semantic_retrieval=semantic_retrieval`` from the ``run_project`` call
inside ``run_portfolio`` → RED."""
factory, recorded = make_recording_client_factory(_ENERGY_REPLY)
store = _tied_pair_store()
project = _bundle_reference_project(tmp_path)
monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: [project])
await run.run_portfolio(
[project.id],
profile="local",
store=store,
client_factory=factory,
top_k=1,
semantic_retrieval=True,
)
gen_prompts = _generation_prompts(recorded)
assert gen_prompts, "the generation call must have happened"
assert any(_TIE_MARKER in p for p in gen_prompts), (
"the cosine-surfaced verdict did not reach the hypothesis prompt — run_portfolio is not "
"forwarding semantic_retrieval to run_project"
)
async def test_run_portfolio_without_the_flag_keeps_the_structural_pick(
make_recording_client_factory, monkeypatch, tmp_path
) -> None:
"""CAUSALITY CONTROL for the portfolio arm — the identical pass with the flag absent must
carry the structural winner and no marker."""
factory, recorded = make_recording_client_factory(_ENERGY_REPLY)
store = _tied_pair_store()
project = _bundle_reference_project(tmp_path)
monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: [project])
await run.run_portfolio(
[project.id],
profile="local",
store=store,
client_factory=factory,
top_k=1,
)
assert all(_TIE_MARKER not in p for p in recorded), (
"the marker reached a prompt without the flag — the portfolio default is not structural"
)