test(s31): close 1 review MAJOR — main()-level load-bearing proof of the --semantic-retrieval wiring
This commit is contained in:
parent
8e9f6603d7
commit
cabe05acf3
1 changed files with 228 additions and 9 deletions
|
|
@ -1,15 +1,35 @@
|
|||
"""Load-bearing CLI test for the observable GoalReached stop (S5.3 Step 3, SC1).
|
||||
"""Load-bearing CLI tests: the observable GoalReached stop (S5.3 Step 3, SC1) and the
|
||||
``--semantic-retrieval`` wiring proof (S3.1 remediation, review finding ``c23e87ee``).
|
||||
|
||||
DETACH THAT TURNS THIS RED: route ``--goals``/``--ledger`` away from ``run_portfolio`` (drop the
|
||||
wiring in ``main()``'s portfolio dispatch) → no goal-stop line reaches stdout → the positive arm's
|
||||
marker assertion fails. The positive+control pair proves the printed scope/øre fields flow from
|
||||
``run_portfolio``'s actual ``GoalReached``/``PortfolioResult`` values — a canned print could not
|
||||
produce BOTH the portfolio-scope and the project-scope variant from the same wiring.
|
||||
DETACH THAT TURNS THE GOAL ARMS RED: route ``--goals``/``--ledger`` away from ``run_portfolio``
|
||||
(drop the wiring in ``main()``'s portfolio dispatch) → no goal-stop line reaches stdout → the
|
||||
positive arm's marker assertion fails. The positive+control pair proves the printed scope/øre
|
||||
fields flow from ``run_portfolio``'s actual ``GoalReached``/``PortfolioResult`` values — a canned
|
||||
print could not produce BOTH the portfolio-scope and the project-scope variant from the same
|
||||
wiring.
|
||||
|
||||
Both arms are provably OFFLINE: the goal is already met before any project builds a chat client
|
||||
DETACH POINTS THAT TURN THE ``--semantic-retrieval`` ARMS RED — two, because the review's finding
|
||||
was that the flag was green-but-dead at exactly these joints:
|
||||
1. Hardcode ``semantic_retrieval=False`` at ``main()``'s full-run ``run_project(...)`` call → the
|
||||
single-project positive loses its marker → RED.
|
||||
2. Drop ``semantic_retrieval=semantic_retrieval`` from the ``run_project(...)`` call inside
|
||||
``run_portfolio`` → the portfolio positive loses its marker → RED.
|
||||
Both were previously covered ONLY below the CLI (``run_project`` driven directly), so either
|
||||
mutation left the suite green. These arms drive ``run.main([...])`` and nothing else.
|
||||
|
||||
Why the fixture looks the way it does: ``main()`` exposes no ``--top-k``, so retrieval returns 3
|
||||
verdicts. The inbox therefore holds FOUR structurally tied verdicts — three decoys whose minted ids
|
||||
sort before the marker's, plus the marker. Structurally (score tied, ``id`` ascending) the marker is
|
||||
the FOURTH and never reaches the prompt; only the cosine term lifts it into the top 3. A two-verdict
|
||||
fixture would put the marker in the prompt either way and the control would prove nothing.
|
||||
|
||||
Both goal arms are provably OFFLINE: the goal is already met before any project builds a chat client
|
||||
(portfolio-hard ``break``s at ``run.py:560-567``; per-project-hard ``continue``s past the only pid at
|
||||
``run.py:571-579`` — no client is ever constructed). Marker value 13731 øre appears nowhere else in
|
||||
the codebase (repo marker convention). No socket/network is exercised (brief NFR).
|
||||
``run.py:571-579`` — no client is ever constructed). The ``--semantic-retrieval`` arms DO run a full
|
||||
pass, but every model call goes to a recording stand-in injected by monkeypatching
|
||||
``run._default_factory`` — the seam ``main()`` resolves through, since it never passes
|
||||
``client_factory``. No socket/network is exercised (brief NFR). Marker value 13731 øre and the
|
||||
``SENTINEL-CLI-45f9a2`` string each appear nowhere else in the codebase (repo marker convention).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -21,9 +41,31 @@ import pytest
|
|||
|
||||
from portfolio_optimiser import run
|
||||
from portfolio_optimiser.ledger import LedgerEntry, SavingsLedger
|
||||
from portfolio_optimiser.reference_domain import Project
|
||||
from portfolio_optimiser.verdicts import ProposalFeatures, capture_verdict, write_verdict
|
||||
|
||||
_MARKER_ORE = 13731 # marker: appears nowhere else in the codebase
|
||||
|
||||
_BUNDLE_DIR = Path(__file__).resolve().parents[1] / "shared" / "examples" / "bygg-energi-mikro"
|
||||
_PID = "BYGG-KONTOR-NORD"
|
||||
|
||||
# The learning signal rides in the RATIONALE — that is what ``ExpeLContextProvider.format_fewshot``
|
||||
# emits into the prompt. It cannot be leaked by bundle context: the string appears nowhere in the
|
||||
# bundle (grep-verified), so its presence in a generation prompt can only have come via the fold.
|
||||
_CLI_MARKER = "SENTINEL-CLI-45f9a2 realiseringsgrad=0.5391"
|
||||
|
||||
# Extra cost codes appended to the bundle's own code. Every combination ties on the structural
|
||||
# score (equal Jaccard, same measure, same magnitude bucket); they differ only in the code set, so
|
||||
# the canonical embedding string differs and cosine has something to separate. Chosen empirically:
|
||||
# the three decoys' minted ids all sort BEFORE the marker's, and the marker's cosine is the highest.
|
||||
_MARKER_CODE = "45.9"
|
||||
_DECOY_CODES = ("58.9", "15.5", "36.8")
|
||||
|
||||
_ENERGY_REPLY = (
|
||||
'{"measure":"LED-retrofit av kontorbelysning","affected_items":'
|
||||
'[{"code":"ENERGI-TOTAL-EL","quantity":300000,"unit_cost":1.0}],"claimed_saving_nok":30000}'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_model_env(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
|
|
@ -85,3 +127,180 @@ def test_per_project_hard_goal_control_distinguishes_scope(tmp_path, capsys) ->
|
|||
assert "scope=project" in out
|
||||
assert "stopped_early=False" in out
|
||||
assert "scope=portfolio" not in out
|
||||
|
||||
|
||||
# --- S3.1: the --semantic-retrieval wiring, proved at main() level (finding c23e87ee) -----------
|
||||
|
||||
|
||||
def _generation_prompts(sink: list[str]) -> list[str]:
|
||||
"""The generation-call prompts (``generate._build_messages`` embeds 'SavingsProposal'),
|
||||
isolated from the debate-round prompts also captured in the shared sink."""
|
||||
return [p for p in sink if "SavingsProposal" in p]
|
||||
|
||||
|
||||
def _tied_inbox(tmp_path: Path) -> str:
|
||||
"""A populated async inbox: three structurally tied decoys plus the marker verdict.
|
||||
|
||||
Written through the real authoring primitives — ``write_verdict(capture_verdict(...))`` — so
|
||||
every id is minted the way the framework mints them, never hand-assigned. ``--verdict-dir`` is
|
||||
the ONLY route by which ``main()`` can hand ``run_project`` a non-empty store."""
|
||||
from portfolio_optimiser.verdicts import bundle_candidate_features
|
||||
|
||||
query = bundle_candidate_features(str(_BUNDLE_DIR))
|
||||
inbox = tmp_path / "inbox"
|
||||
|
||||
def tied(extra_code: str, decision: str, rationale: str) -> None:
|
||||
write_verdict(
|
||||
str(inbox),
|
||||
capture_verdict(
|
||||
ProposalFeatures(
|
||||
affected_codes=query.affected_codes | {extra_code},
|
||||
measure_type=query.measure_type,
|
||||
claimed_saving_nok=query.claimed_saving_nok,
|
||||
# ``description == measure`` is the shape both live minting paths emit.
|
||||
description=query.measure_type,
|
||||
),
|
||||
decision,
|
||||
rationale,
|
||||
),
|
||||
)
|
||||
|
||||
for code in _DECOY_CODES:
|
||||
tied(code, "rejected", f"ingen realiseringsdata for tiltak {code}")
|
||||
tied(_MARKER_CODE, "approved", f"tidligere LED-dom [{_CLI_MARKER}]")
|
||||
return str(inbox)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def _recorded_main(monkeypatch):
|
||||
"""Inject a recording chat client into the seam ``main()`` actually resolves through.
|
||||
|
||||
``main()`` never passes ``client_factory``, so ``run_project`` falls back to the module-level
|
||||
``_default_factory``. Patching it is therefore the only observation point available to a
|
||||
main()-level test — and patching it (rather than passing a factory) is what keeps this test
|
||||
honest about the CLI path."""
|
||||
from tests.conftest import _RecordingChatClient
|
||||
|
||||
sink: list[str] = []
|
||||
monkeypatch.setattr(
|
||||
"portfolio_optimiser.run._default_factory",
|
||||
lambda profile: lambda role: _RecordingChatClient(sink, _ENERGY_REPLY),
|
||||
)
|
||||
return sink
|
||||
|
||||
|
||||
def test_cli_semantic_retrieval_reaches_the_hypothesis_prompt(tmp_path, _recorded_main) -> None:
|
||||
"""POSITIVE — ``run.main([... --semantic-retrieval])`` must carry the cosine-surfaced verdict's
|
||||
rationale into the hypothesis prompt.
|
||||
|
||||
This is the arm the review's ``c23e87ee`` demanded: every prior proof drove ``run_project``
|
||||
directly, so the CLI could stop forwarding the flag without a single test noticing.
|
||||
|
||||
Detach point: hardcode ``semantic_retrieval=False`` at ``main()``'s ``run_project(...)`` call
|
||||
→ RED."""
|
||||
rc = run.main(
|
||||
[
|
||||
_PID,
|
||||
"--docs-dir",
|
||||
str(_BUNDLE_DIR),
|
||||
"--bundle-dir",
|
||||
str(_BUNDLE_DIR),
|
||||
"--verdict-dir",
|
||||
_tied_inbox(tmp_path),
|
||||
"--semantic-retrieval",
|
||||
]
|
||||
)
|
||||
assert rc == 0
|
||||
|
||||
gen_prompts = _generation_prompts(_recorded_main)
|
||||
assert gen_prompts, "the generation call must have happened"
|
||||
assert any(_CLI_MARKER in p for p in gen_prompts), (
|
||||
"the cosine-surfaced verdict did not reach the hypothesis prompt — main() is not "
|
||||
"forwarding --semantic-retrieval into the Step-1 fold"
|
||||
)
|
||||
|
||||
|
||||
def test_cli_without_the_flag_leaves_the_marker_out(tmp_path, _recorded_main) -> None:
|
||||
"""CAUSALITY CONTROL — the identical CLI invocation WITHOUT the flag must not carry the marker.
|
||||
|
||||
Without this the positive proves nothing: it would pass merely because the inbox contains the
|
||||
marker. Structurally the marker is the fourth of four tied verdicts and retrieval returns three,
|
||||
so only the cosine term can reach it."""
|
||||
rc = run.main(
|
||||
[
|
||||
_PID,
|
||||
"--docs-dir",
|
||||
str(_BUNDLE_DIR),
|
||||
"--bundle-dir",
|
||||
str(_BUNDLE_DIR),
|
||||
"--verdict-dir",
|
||||
_tied_inbox(tmp_path),
|
||||
]
|
||||
)
|
||||
assert rc == 0
|
||||
|
||||
assert all(_CLI_MARKER not in p for p in _recorded_main), (
|
||||
"the marker reached a prompt with --semantic-retrieval OFF — the default CLI path is not "
|
||||
"the structural ranking, so the positive arm is not load-bearing"
|
||||
)
|
||||
|
||||
|
||||
def test_cli_portfolio_mode_forwards_semantic_retrieval(tmp_path, monkeypatch, _recorded_main):
|
||||
"""POSITIVE (portfolio arm) — ``main(["--portfolio", "--semantic-retrieval"])`` must reach each
|
||||
project's Step-1 fold.
|
||||
|
||||
Portfolio mode refuses ``--verdict-dir`` (it is single-project-only), so the inbox arrives on
|
||||
the ``Project`` record instead — the same topology
|
||||
``tests/test_portfolio_learning_loadbearing.py`` uses.
|
||||
|
||||
Detach point: drop ``semantic_retrieval=semantic_retrieval`` from the ``run_project(...)`` call
|
||||
inside ``run_portfolio`` → RED."""
|
||||
project = _bundle_project(tmp_path, verdict_dir=_tied_inbox(tmp_path))
|
||||
monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (project,))
|
||||
|
||||
rc = run.main(["--portfolio", "--semantic-retrieval"])
|
||||
assert rc == 0
|
||||
|
||||
gen_prompts = _generation_prompts(_recorded_main)
|
||||
assert gen_prompts, "the generation call must have happened"
|
||||
assert any(_CLI_MARKER in p for p in gen_prompts), (
|
||||
"the marker did not reach the hypothesis prompt in portfolio mode — run_portfolio is not "
|
||||
"forwarding semantic_retrieval to run_project"
|
||||
)
|
||||
|
||||
|
||||
def test_cli_portfolio_mode_without_the_flag_leaves_the_marker_out(
|
||||
tmp_path, monkeypatch, _recorded_main
|
||||
):
|
||||
"""CAUSALITY CONTROL for the portfolio arm."""
|
||||
project = _bundle_project(tmp_path, verdict_dir=_tied_inbox(tmp_path))
|
||||
monkeypatch.setattr("portfolio_optimiser.run.load_reference_projects", lambda: (project,))
|
||||
|
||||
rc = run.main(["--portfolio"])
|
||||
assert rc == 0
|
||||
|
||||
assert all(_CLI_MARKER not in p for p in _recorded_main), (
|
||||
"the marker reached a prompt without the flag — the portfolio default is not structural"
|
||||
)
|
||||
|
||||
|
||||
def _bundle_project(tmp_path: Path, *, verdict_dir: str) -> Project:
|
||||
"""One bundle-backed project pointed at the SAME bundle the tied set was derived from — a
|
||||
different bundle would yield a different query and untie the set."""
|
||||
docs = tmp_path / "portfolio-docs"
|
||||
docs.mkdir(exist_ok=True)
|
||||
(docs / "cost.txt").write_text(
|
||||
"LED-retrofit av lysrorarmaturer i kontorlokaler reduserte energikostnaden.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return Project(
|
||||
id=_PID,
|
||||
name="Bundle-backed portfolio project",
|
||||
description="bundle-backed project for the CLI semantic-retrieval forwarding proof",
|
||||
currency="NOK",
|
||||
cost_items=(),
|
||||
docs_dir=str(docs),
|
||||
verdict_input={"decision": "approved", "rationale": "expert reviewed (sim)"},
|
||||
bundle_dir=str(_BUNDLE_DIR),
|
||||
verdict_dir=verdict_dir,
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue